Merge branch 'upstream' into concedo_experimental

# Conflicts:
#	ggml/src/ggml-opencl/ggml-opencl.cpp
#	tests/test-backend-ops.cpp
#	tests/test-quantize-fns.cpp
#	tools/server/README.md
#	tools/ui/src/lib/constants/settings-registry.ts
This commit is contained in:
Concedo
2026-07-11 09:20:40 +08:00
103 changed files with 2633 additions and 1170 deletions
+1
View File
@@ -32,6 +32,7 @@ struct quant_option {
static const std::vector<quant_option> QUANT_OPTIONS = {
{ "Q1_0", LLAMA_FTYPE_MOSTLY_Q1_0, " 1.125 bpw quantization", },
{ "Q2_0", LLAMA_FTYPE_MOSTLY_Q2_0, " 2.25 bpw quantization (group 64)", },
{ "Q4_0", LLAMA_FTYPE_MOSTLY_Q4_0, " 4.34G, +0.4685 ppl @ Llama-3-8B", },
{ "Q4_1", LLAMA_FTYPE_MOSTLY_Q4_1, " 4.78G, +0.4511 ppl @ Llama-3-8B", },
{ "MXFP4_MOE",LLAMA_FTYPE_MOSTLY_MXFP4_MOE," MXFP4 MoE", },
+6 -4
View File
@@ -57,7 +57,7 @@ The core architecture consists of the following components:
- `server_tokens`: Unified representation of token sequences (supports both text and multimodal tokens); used by `server_task` and `server_slot`.
- `server_prompt_checkpoint`: For recurrent (e.g., RWKV) and SWA models, stores snapshots of KV cache state. Enables reuse when subsequent requests share the same prompt prefix, saving redundant computation.
- `server_models`: Standalone component for managing multiple backend instances (used in router mode). It is completely independent of `server_context`.
- `stream_session_manager`: Process wide owner of resumable SSE stream sessions (`g_stream_sessions`), keyed by conversation id. Backs the replay buffer that lets a client reattach to a generation after an HTTP disconnect. See the "Resumable streaming" section below.
- `stream_session_manager`: process wide owner of resumable SSE stream sessions, keyed by conversation id. A file-static singleton inside `server-stream.cpp`, driven through `server_stream_session_manager_start/stop`. Backs the replay buffer that lets a client reattach to a generation after an HTTP disconnect. See the "Resumable streaming" section below.
```mermaid
graph TD
@@ -127,10 +127,12 @@ It is opt in via the `X-Conversation-Id` header on `POST /v1/chat/completions`.
The feature lives entirely in `server-stream.{h,cpp}` and rests on three types:
- `stream_session`: a bounded ring buffer (4 MiB cap, oldest bytes drop first) plus a condvar. `append` pushes raw SSE bytes, `read_from` drains from any offset and blocks for live bytes or finalize, `finalize` wakes readers, `cancel` stops the producer. One conv maps to at most one live session.
- `stream_session_manager` (`g_stream_sessions`): owns all sessions keyed by conv id, enforces the one conv one session invariant via `create_or_replace`, and runs a GC thread that drops completed sessions past their TTL.
- `stream_session_manager`: a file-static singleton (`g_stream_sessions`) inside `server-stream.cpp`, owns all sessions keyed by conv id, enforces the one conv one session invariant via `create_or_replace`, and runs a GC thread that drops completed sessions past their TTL. Exposed to main only through `server_stream_session_manager_start/stop`.
- `stream_pipe_producer` / `stream_pipe_consumer`: the write and read ends. The producer owns the session lifetime and finalizes it on destruction; the consumer is read only and never finalizes, so a reader detaching cannot kill a running generation.
Producer side: `server_res_generator` attaches a producer pipe when the header is present. The HTTP content provider mirrors every chunk into the ring before writing it to the socket. While a pipe is attached, `stream_aware_should_stop` ignores peer disconnect, so a dropped socket does not stop generation: only an explicit `DELETE` does. When the peer leaves early, `on_complete` calls `close()`, which drains the rest of the generation into the ring on the http worker.
The implementation is hidden in `server-stream.cpp` (pimpl). The header exposes only the route handler factories, `server_stream_session_attach_pipe`, `server_stream_aware_should_stop`, `server_stream_conv_id_from_headers` and the GC lifecycle; the session, manager and consumer types stay in the `.cpp`.
Producer side: `server_res_generator` attaches a producer pipe when the header is present. The HTTP content provider mirrors every chunk into the ring before writing it to the socket. While a pipe is attached, `server_stream_aware_should_stop` ignores peer disconnect, so a dropped socket does not stop generation: only an explicit `DELETE` does. When the peer leaves early, `on_complete` calls `close()`, which drains the rest of the generation into the ring on the http worker.
Lifetime safety: the producer pipe holds a shared `alive` flag also captured by the session cancel hook. `~server_res_generator` calls `cleanup()` to clear that hook while the reader is still alive, so a `cancel` arriving during teardown can never call `stop()` on a freed response. This ordering is the most fragile part of the feature: finalizing or destroying the producer before `cleanup()` runs reintroduces a use after free.
@@ -144,7 +146,7 @@ Routes:
Router mode binds the same paths to proxy handlers. A `conv_id -> child` map (`conv_models`), populated when a POST is routed, resolves the owning child in one lookup with no polling. The lookup groups ids per child; GET and DELETE proxy straight to the owner. This loopback REST hop is expected to move to a websocket IPC later, swapping only the transport.
Lifecycle: `g_stream_sessions.start_gc()` runs in main after common init, `stop_gc()` runs first in `clean_up()` and finalizes every live session so no reader hangs. Reader blocking and the post drop drain both run on httplib worker threads, which block on a condvar rather than spin.
Lifecycle: `server_stream_session_manager_start()` runs in main after common init, `server_stream_session_manager_stop()` runs first in `clean_up()` and finalizes every live session so no reader hangs. Reader blocking and the post drop drain both run on httplib worker threads, which block on a condvar rather than spin.
| Constant | Value | Role |
| --- | --- | --- |
+56 -111
View File
@@ -897,8 +897,10 @@ private:
server_batch batch;
llama_model_ptr model_dft;
llama_context_ptr ctx_dft;
llama_model * model_dft = nullptr;
llama_context * ctx_dft = nullptr;
common_speculative_init_result_ptr spec_init;
common_context_seq_rm_type ctx_tgt_seq_rm_type = COMMON_CONTEXT_SEQ_RM_TYPE_NO;
common_context_seq_rm_type ctx_dft_seq_rm_type = COMMON_CONTEXT_SEQ_RM_TYPE_NO;
@@ -939,8 +941,10 @@ private:
void destroy() {
spec.reset();
ctx_dft.reset();
model_dft.reset();
spec_init.reset();
ctx_dft = nullptr;
model_dft = nullptr;
llama_init.reset();
@@ -1084,30 +1088,15 @@ private:
// optionally reserve VRAM for the draft / MTP context before fitting the target model
if (params_base.fit_params) {
if (has_spec) {
common_params params_dft = params_base;
bool measure_model_bytes = true;
// MTP draft context lives on the target model, only context+compute are new
bool measure_model_bytes = has_draft;
if (has_draft) {
const auto & params_spec = params_base.speculative.draft;
params_dft.devices = params_spec.devices;
params_dft.model = params_spec.mparams;
params_dft.n_gpu_layers = params_spec.n_gpu_layers;
params_dft.cache_type_k = params_spec.cache_type_k;
params_dft.cache_type_v = params_spec.cache_type_v;
params_dft.tensor_buft_overrides = params_spec.tensor_buft_overrides;
} else {
// MTP draft context lives on the target model, only context+compute are new
measure_model_bytes = false;
}
params_dft.n_outputs_max = params_base.n_parallel;
common_params params_dft = common_base_params_to_speculative(params_base);
auto mparams_dft = common_model_params_to_llama(params_dft);
auto cparams_dft = common_context_params_to_llama(params_dft);
if (spec_mtp) {
cparams_dft.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
cparams_dft.type_k = params_base.speculative.draft.cache_type_k;
cparams_dft.type_v = params_base.speculative.draft.cache_type_v;
}
cparams_dft.n_rs_seq = 0;
@@ -1175,82 +1164,36 @@ private:
add_bos_token = llama_vocab_get_add_bos(vocab);
if (has_draft) {
// TODO speculative: move to common/speculative.cpp?
const auto & params_spec = params_base.speculative.draft;
SRV_TRC("loading draft model '%s'\n", params_spec.mparams.path.c_str());
auto params_dft = params_base;
params_dft.devices = params_spec.devices;
params_dft.model = params_spec.mparams;
params_dft.n_gpu_layers = params_spec.n_gpu_layers;
params_dft.cache_type_k = params_spec.cache_type_k;
params_dft.cache_type_v = params_spec.cache_type_v;
if (params_spec.cpuparams.n_threads > 0) {
params_dft.cpuparams.n_threads = params_spec.cpuparams.n_threads;
params_dft.cpuparams_batch.n_threads = params_spec.cpuparams_batch.n_threads;
}
params_dft.tensor_buft_overrides = params_spec.tensor_buft_overrides;
auto mparams_dft = common_model_params_to_llama(params_dft);
// progress callback
mparams_dft.progress_callback = load_progress_callback;
mparams_dft.progress_callback_user_data = &load_progress_spec;
model_dft.reset(llama_model_load_from_file(params_dft.model.path.c_str(), mparams_dft));
if (model_dft == nullptr) {
SRV_ERR("failed to load draft model, '%s'\n", params_dft.model.path.c_str());
return false;
}
auto cparams = common_context_params_to_llama(params_dft);
if (spec_mtp) {
cparams.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
}
// note: for small models maybe we can set this to the maximum possible draft from all speculative types
// the extra memory for small models is likely negligible?
cparams.n_rs_seq = 0;
cparams.ctx_other = ctx_tgt;
ctx_dft.reset(llama_init_from_model(model_dft.get(), cparams));
if (ctx_dft == nullptr) {
SRV_ERR("%s", "failed to create draft context\n");
return false;
}
params_base.speculative.draft.ctx_tgt = ctx_tgt;
params_base.speculative.draft.ctx_dft = ctx_dft.get();
} else if (spec_mtp) {
// no new model load, so we simply report 0.0 and 1.0 progress
if (has_spec) {
// spec_mtp doesn't use load a model internally, so we report 0.0 and 1.0 manually
load_progress_callback(0.0f, &load_progress_spec);
load_progress_spec.t_last_load_progress_ms = 0; // reset so internal cbs aren't delayed
SRV_TRC("creating MTP draft context against the target model '%s'\n",
params_base.model.path.c_str());
{
common_params params_dft = common_base_params_to_speculative(params_base);
auto cparams_mtp = common_context_params_to_llama(params_base);
cparams_mtp.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
cparams_mtp.type_k = params_base.speculative.draft.cache_type_k;
cparams_mtp.type_v = params_base.speculative.draft.cache_type_v;
cparams_mtp.n_rs_seq = 0;
cparams_mtp.n_outputs_max = params_base.n_parallel;
cparams_mtp.ctx_other = ctx_tgt;
// progress callback
params_dft.load_progress_callback = load_progress_callback;
params_dft.load_progress_callback_user_data = &load_progress_spec;
ctx_dft.reset(llama_init_from_model(model_tgt, cparams_mtp));
if (ctx_dft == nullptr) {
SRV_ERR("%s", "failed to create MTP context\n");
return false;
spec_init = common_speculative_init_from_params(params_dft, model_tgt, ctx_tgt);
model_dft = spec_init->model();
ctx_dft = spec_init->context();
if (has_draft && model_dft == nullptr) {
SRV_ERR("failed to load draft model, '%s'\n", params_dft.model.path.c_str());
return false;
}
if (ctx_dft == nullptr) {
SRV_ERR("%s", "failed to create MTP context\n");
return false;
}
params_base.speculative.draft.ctx_tgt = ctx_tgt;
params_base.speculative.draft.ctx_dft = ctx_dft;
}
params_base.speculative.draft.ctx_tgt = ctx_tgt;
params_base.speculative.draft.ctx_dft = ctx_dft.get();
load_progress_callback(1.0f, &load_progress_spec);
}
@@ -1343,13 +1286,15 @@ private:
}
if (ctx_dft) {
ctx_dft_seq_rm_type = common_context_can_seq_rm(ctx_dft.get());
ctx_dft_seq_rm_type = common_context_can_seq_rm(ctx_dft);
}
if (spec) {
SRV_TRC("%s", "speculative decoding context initialized\n");
} else {
ctx_dft.reset();
spec_init.reset();
ctx_dft = nullptr;
model_dft = nullptr;
}
for (int i = 0; i < params_base.n_parallel; i++) {
@@ -1357,7 +1302,7 @@ private:
slot.id = i;
slot.ctx_tgt = ctx_tgt;
slot.ctx_dft = ctx_dft.get();
slot.ctx_dft = ctx_dft;
slot.spec = spec.get();
slot.n_ctx = n_ctx_slot;
@@ -2362,8 +2307,8 @@ private:
// this is not true for SWA models: https://github.com/ggml-org/llama.cpp/pull/24411#issuecomment-4677983225
cur.update_pos(slot.prompt.n_tokens() - n_tokens_cur, pos_min, pos_max);
cur.update_tgt(ctx_tgt, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
cur.update_dft(ctx_dft.get(), slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
cur.update_tgt(ctx_tgt, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
cur.update_dft(ctx_dft, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
// stash the draft's speculative state with the checkpoint
common_speculative_get_state(spec.get(), slot.id, cur.data_spec);
@@ -2899,8 +2844,8 @@ private:
common_context_seq_add(ctx_tgt, slot.id, n_keep + n_discard, slot.prompt.n_tokens(), -n_discard);
if (ctx_dft) {
common_context_seq_rm (ctx_dft.get(), slot.id, n_keep , n_keep + n_discard);
common_context_seq_add(ctx_dft.get(), slot.id, n_keep + n_discard, slot.prompt.tokens.pos_next(), -n_discard);
common_context_seq_rm (ctx_dft, slot.id, n_keep , n_keep + n_discard);
common_context_seq_add(ctx_dft, slot.id, n_keep + n_discard, slot.prompt.tokens.pos_next(), -n_discard);
}
// add generated tokens to cache
@@ -2972,7 +2917,7 @@ private:
llama_memory_seq_pos_max(llama_get_memory(ctx_tgt), slot.id));
if (use_ckpt_dft) {
slot.spec_ckpt.update_dft(ctx_dft.get(), slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
slot.spec_ckpt.update_dft(ctx_dft, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
}
slot.spec_prompt = slot.prompt.tokens.get_text_tokens();
@@ -3009,10 +2954,10 @@ private:
if (ctx_dft) {
if (use_ckpt_dft) {
ckpt.load_dft(ctx_dft.get(), slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
ckpt.load_dft(ctx_dft, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
}
common_context_seq_rm(ctx_dft.get(), slot.id, ckpt.pos_max + 1, -1);
common_context_seq_rm(ctx_dft, slot.id, ckpt.pos_max + 1, -1);
}
if (!draft.empty()) {
@@ -3021,7 +2966,7 @@ private:
(ctx_tgt_seq_rm_type == COMMON_CONTEXT_SEQ_RM_TYPE_RS && draft.size() > llama_n_rs_seq(ctx_tgt));
const bool use_ckpt_dft =
(ctx_dft_seq_rm_type == COMMON_CONTEXT_SEQ_RM_TYPE_RS && draft.size() > llama_n_rs_seq(ctx_dft.get()));
(ctx_dft_seq_rm_type == COMMON_CONTEXT_SEQ_RM_TYPE_RS && draft.size() > llama_n_rs_seq(ctx_dft));
if (use_ckpt_tgt) {
//const int64_t t_start = ggml_time_us();
@@ -3038,7 +2983,7 @@ private:
}
if (use_ckpt_dft) {
ckpt.update_dft(ctx_dft.get(), slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
ckpt.update_dft(ctx_dft, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
}
}
});
@@ -3219,8 +3164,8 @@ private:
common_context_seq_add(ctx_tgt, slot.id, head_c, head_c + n_match, kv_shift);
if (ctx_dft) {
common_context_seq_rm (ctx_dft.get(), slot.id, head_p, head_c);
common_context_seq_add(ctx_dft.get(), slot.id, head_c, head_c + n_match, kv_shift);
common_context_seq_rm (ctx_dft, slot.id, head_p, head_c);
common_context_seq_add(ctx_dft, slot.id, head_c, head_c + n_match, kv_shift);
}
for (size_t i = 0; i < n_match; i++) {
@@ -3320,8 +3265,8 @@ private:
if (!do_reset) {
// restore the context checkpoint
it->load_tgt(ctx_tgt, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
it->load_dft(ctx_dft.get(), slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
it->load_tgt(ctx_tgt, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
it->load_dft(ctx_dft, slot.id, LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY);
// restore the draft's speculative state
common_speculative_set_state(spec.get(), slot.id, it->data_spec);
@@ -3395,7 +3340,7 @@ private:
common_context_seq_rm(ctx_tgt, slot.id, p0, -1);
if (ctx_dft) {
common_context_seq_rm(ctx_dft.get(), slot.id, p0, -1);
common_context_seq_rm(ctx_dft, slot.id, p0, -1);
}
// If using an alora, there may be uncached tokens that come
@@ -4243,7 +4188,7 @@ std::unique_ptr<server_res_generator> server_routes::handle_completions_impl(
}
};
auto effective_should_stop = stream_aware_should_stop(res_this, req.should_stop);
auto effective_should_stop = server_stream_aware_should_stop(res_this, req.should_stop);
try {
if (effective_should_stop()) {
@@ -4339,7 +4284,7 @@ std::unique_ptr<server_res_generator> server_routes::handle_completions_impl(
// attach a producer pipe to the response when X-Conversation-Id is present.
// the pipe mirrors SSE chunks into the ring buffer and wires up the cancel hook.
stream_session_attach_pipe(*res, req.headers);
server_stream_session_attach_pipe(*res, req.headers);
return res;
}
+2 -2
View File
@@ -1681,7 +1681,7 @@ void server_models_routes::init_routes() {
}
// remember which child serves this conversation so the stream routes can route straight
// to it without polling, keyed on the exact conv id from the header
std::string conv_id = stream_conv_id_from_headers(req.headers);
std::string conv_id = server_stream_conv_id_from_headers(req.headers);
if (!conv_id.empty()) {
models.conv_models.remember(conv_id, name);
}
@@ -1896,7 +1896,7 @@ void server_models_routes::init_routes() {
if (!from.empty()) {
child_path += "?from=" + from;
}
SRV_INF("proxying stream resume to model %s on port %d, path=%s\n",
SRV_TRC("proxying stream resume to model %s on port %d, path=%s\n",
owner->name.c_str(), owner->port, child_path.c_str());
auto proxy = std::make_unique<server_http_proxy>(
"GET",
+147 -41
View File
@@ -6,6 +6,12 @@
#include <chrono>
#include <memory>
#include <utility>
#include <shared_mutex>
enum class stream_read_status {
OK,
OFFSET_LOST,
};
namespace {
constexpr int64_t STREAM_SESSION_TTL_SECONDS = 300;
@@ -13,7 +19,6 @@ constexpr size_t STREAM_SESSION_MAX_BYTES = 4 * 1024 * 1024;
constexpr int64_t STREAM_SESSION_GC_INTERVAL_SECONDS = 60;
constexpr int64_t STREAM_READ_WAKE_INTERVAL_MS = 200;
// returns unix time in seconds
int64_t now_seconds() {
return std::chrono::duration_cast<std::chrono::seconds>(
std::chrono::system_clock::now().time_since_epoch()
@@ -21,6 +26,91 @@ int64_t now_seconds() {
}
}
// owns all live sessions keyed by conversation_id, one conv = at most one live session.
// a periodic GC evicts expired ones
class stream_session_manager {
public:
stream_session_manager();
~stream_session_manager();
stream_session_manager(const stream_session_manager &) = delete;
stream_session_manager & operator=(const stream_session_manager &) = delete;
// install a new session, evicting and cancelling any previous one. conversation_id must be non empty
stream_session_ptr create_or_replace(const std::string & conversation_id);
stream_session_ptr get(const std::string & conversation_id);
std::vector<stream_session_ptr> list_all() const;
void evict(const std::string & conversation_id);
void evict_and_cancel(const std::string & conversation_id);
void start_gc();
void stop_gc();
private:
void gc_loop();
mutable std::shared_mutex map_mu;
std::unordered_map<std::string, stream_session_ptr> sessions; // key: conversation_id
std::thread gc_thread;
bool running;
std::mutex gc_wake_mu;
std::condition_variable gc_wake_cv;
};
// process wide manager, lifecycle controlled by llama-server main() via start_gc/stop_gc
static stream_session_manager g_stream_sessions;
void server_stream_session_manager_start() {
g_stream_sessions.start_gc();
}
void server_stream_session_manager_stop() {
g_stream_sessions.stop_gc();
}
struct stream_session {
std::string conversation_id;
int64_t started_ts; // unix seconds at construction
stream_session(std::string conversation_id_, size_t max_bytes_);
stream_session(const stream_session &) = delete;
stream_session & operator=(const stream_session &) = delete;
bool append(const char * data, size_t len);
void finalize();
// drain from offset into sink, blocking for more bytes or finalize. OFFSET_LOST if offset
// fell below the dropped prefix
stream_read_status read_from(size_t offset,
const std::function<bool(const char *, size_t)> & sink,
const std::function<bool()> & should_stop);
bool is_done() const;
bool is_cancelled() const;
size_t total_size() const; // bytes that ever entered the session
size_t dropped_prefix() const; // bytes evicted from the front due to cap
int64_t completed_at() const; // 0 while alive, unix seconds after finalize
void set_stop_producer(std::function<void()> fn);
void cancel();
private:
mutable std::mutex mu;
std::condition_variable cv;
std::vector<char> buffer;
size_t prefix_dropped;
size_t cap_bytes;
bool done;
std::atomic<bool> cancelled; // polled lock-free by the should_stop closure, no mu
int64_t completed_ts;
std::function<void()> stop_producer;
};
stream_session::stream_session(std::string conversation_id_, size_t max_bytes_)
: conversation_id(std::move(conversation_id_))
, started_ts(now_seconds())
@@ -38,7 +128,7 @@ bool stream_session::append(const char * data, size_t len) {
}
{
std::lock_guard<std::mutex> lock(mu);
if (done.load(std::memory_order_relaxed)) {
if (done) {
return false;
}
if (len >= cap_bytes) {
@@ -62,11 +152,14 @@ bool stream_session::append(const char * data, size_t len) {
}
void stream_session::finalize() {
bool was_done = done.exchange(true, std::memory_order_acq_rel);
if (was_done) {
return;
{
std::lock_guard<std::mutex> lock(mu);
if (done) {
return;
}
done = true;
completed_ts = now_seconds();
}
completed_ts.store(now_seconds(), std::memory_order_release);
cv.notify_all();
}
@@ -96,7 +189,7 @@ stream_read_status stream_session::read_from(size_t offset,
lock.lock();
continue;
}
if (done.load(std::memory_order_acquire)) {
if (done) {
return stream_read_status::OK;
}
// wait for new bytes, finalize, or a periodic wake to re check should_stop
@@ -105,7 +198,8 @@ stream_read_status stream_session::read_from(size_t offset,
}
bool stream_session::is_done() const {
return done.load(std::memory_order_acquire);
std::lock_guard<std::mutex> lock(mu);
return done;
}
size_t stream_session::total_size() const {
@@ -119,7 +213,8 @@ size_t stream_session::dropped_prefix() const {
}
int64_t stream_session::completed_at() const {
return completed_ts.load(std::memory_order_acquire);
std::lock_guard<std::mutex> lock(mu);
return completed_ts;
}
void stream_session::set_stop_producer(std::function<void()> fn) {
@@ -128,7 +223,7 @@ void stream_session::set_stop_producer(std::function<void()> fn) {
}
void stream_session::cancel() {
// flip cancelled first so the producer-side stream_aware_should_stop can break out of the
// flip cancelled first so the producer-side server_stream_aware_should_stop can break out of the
// recv() wait even if remove_waiting_task_ids does not notify the condvar (the cancel task
// posted by rd.stop() will eventually notify, but we do not want to depend on that timing)
cancelled.store(true, std::memory_order_release);
@@ -237,18 +332,24 @@ void stream_session_manager::evict_and_cancel(const std::string & conversation_i
}
void stream_session_manager::start_gc() {
if (running.exchange(true)) {
return;
{
std::lock_guard<std::mutex> lock(gc_wake_mu);
if (running) {
return;
}
running = true;
}
gc_thread = std::thread([this] { gc_loop(); });
}
void stream_session_manager::stop_gc() {
bool was_running = running.exchange(false);
bool was_running;
{
std::lock_guard<std::mutex> lock(gc_wake_mu);
was_running = running;
running = false;
}
if (was_running) {
{
std::lock_guard<std::mutex> lock(gc_wake_mu);
}
gc_wake_cv.notify_all();
if (gc_thread.joinable()) {
gc_thread.join();
@@ -270,15 +371,15 @@ void stream_session_manager::stop_gc() {
}
void stream_session_manager::gc_loop() {
while (running.load(std::memory_order_acquire)) {
while (true) {
{
std::unique_lock<std::mutex> lock(gc_wake_mu);
gc_wake_cv.wait_for(lock,
std::chrono::seconds(STREAM_SESSION_GC_INTERVAL_SECONDS),
[this] { return !running.load(std::memory_order_acquire); });
}
if (!running.load(std::memory_order_acquire)) {
return;
[this] { return !running; });
if (!running) {
return;
}
}
int64_t cutoff = now_seconds() - STREAM_SESSION_TTL_SECONDS;
std::vector<stream_session_ptr> to_drop;
@@ -301,10 +402,19 @@ void stream_session_manager::gc_loop() {
}
}
// process wide manager, lifecycle controlled by llama-server main() via start_gc/stop_gc
stream_session_manager g_stream_sessions;
// stream_pipe
// stream_pipe ---------------------------------------------------------------------------------
// consumer end: read-only replay of the ring buffer, the destructor does not finalize the session
struct stream_pipe_consumer : stream_pipe {
stream_read_status read(size_t & offset,
const std::function<bool(const char *, size_t)> & sink,
const std::function<bool()> & should_stop);
static std::shared_ptr<stream_pipe_consumer> create(stream_session_ptr session);
private:
explicit stream_pipe_consumer(stream_session_ptr session);
};
stream_pipe::stream_pipe(stream_session_ptr session)
: session_(std::move(session)) {
@@ -408,12 +518,10 @@ static server_http_res_ptr make_error_response(int status, const std::string & m
return res;
}
server_http_context::handler_t make_stream_get_handler() {
server_http_context::handler_t server_stream_make_get_handler() {
return [](const server_http_req & req) -> server_http_res_ptr {
// GET /v1/stream/<conv_id>?from=N replays the SSE bytes already buffered for the
// session, blocks for more bytes when the session is still running, returns when
// the session is finalized. the body is streamed back as text/event-stream so the
// browser EventSource can attach to it like a fresh request
// GET /v1/stream/<conv_id>?from=N replays buffered SSE bytes then blocks for live
// bytes until the session finalizes, streamed as text/event-stream for EventSource
std::string conv_id = req.get_param("conv_id");
if (conv_id.empty()) {
return make_error_response(400, "Missing conversation id in path", ERROR_TYPE_INVALID_REQUEST);
@@ -459,11 +567,10 @@ server_http_context::handler_t make_stream_get_handler() {
};
}
server_http_context::handler_t make_streams_lookup_handler() {
server_http_context::handler_t server_stream_make_lookup_handler() {
return [](const server_http_req & req) -> server_http_res_ptr {
// POST /v1/streams/lookup with body {"conversation_ids": ["X", "Y", ...]} returns the
// matching sessions, only for ids the caller already knows. each id matches the exact key
// and any "<id>::<model>" variant, so one lookup covers every per model session for a conv
// POST /v1/streams/lookup returns the matching sessions, only for ids the caller already
// knows. each id matches the exact key and any "<id>::<model>" per model variant
std::vector<std::string> requested;
try {
json body = json::parse(req.body);
@@ -518,11 +625,10 @@ server_http_context::handler_t make_streams_lookup_handler() {
};
}
server_http_context::handler_t make_stream_delete_handler() {
server_http_context::handler_t server_stream_make_delete_handler() {
return [](const server_http_req & req) -> server_http_res_ptr {
// DELETE /v1/stream/<conv_id> is the explicit user Stop, cancels the producer hook
// wired by handle_completions_impl and evicts the buffer. idempotent, a session that
// already finalized or was never created returns 204 either way
// DELETE /v1/stream/<conv_id> is the explicit user Stop, cancels the producer and evicts
// the buffer. idempotent, returns 204 even if the session was already gone
std::string conv_id = req.get_param("conv_id");
if (conv_id.empty()) {
return make_error_response(400, "Missing conversation id in path", ERROR_TYPE_INVALID_REQUEST);
@@ -536,7 +642,7 @@ server_http_context::handler_t make_stream_delete_handler() {
};
}
std::string stream_conv_id_from_headers(const std::map<std::string, std::string> & headers) {
std::string server_stream_conv_id_from_headers(const std::map<std::string, std::string> & headers) {
// case-insensitive scan for x-conversation-id
static constexpr char target[] = "x-conversation-id";
static constexpr size_t target_len = sizeof(target) - 1;
@@ -555,8 +661,8 @@ std::string stream_conv_id_from_headers(const std::map<std::string, std::string>
return std::string();
}
void stream_session_attach_pipe(server_http_res & res, const std::map<std::string, std::string> & headers) {
std::string conversation_id = stream_conv_id_from_headers(headers);
void server_stream_session_attach_pipe(server_http_res & res, const std::map<std::string, std::string> & headers) {
std::string conversation_id = server_stream_conv_id_from_headers(headers);
SRV_TRC("conv_id=%s (empty=%d)\n", conversation_id.c_str(), conversation_id.empty() ? 1 : 0);
if (conversation_id.empty()) {
return;
@@ -565,7 +671,7 @@ void stream_session_attach_pipe(server_http_res & res, const std::map<std::strin
res.spipe = stream_pipe_producer::create(session, res);
}
std::function<bool()> stream_aware_should_stop(server_http_res * res, std::function<bool()> fallback) {
std::function<bool()> server_stream_aware_should_stop(server_http_res * res, std::function<bool()> fallback) {
return [res, fallback = std::move(fallback)]() -> bool {
if (res->spipe) {
return res->spipe->is_cancelled();
+15 -136
View File
@@ -3,81 +3,23 @@
#include "server-http.h"
#include <atomic>
#include <condition_variable>
#include <cstddef>
#include <cstdint>
#include <functional>
#include <memory>
#include <mutex>
#include <shared_mutex>
#include <string>
#include <thread>
#include <unordered_map>
#include <vector>
enum class stream_read_status {
OK,
OFFSET_LOST,
};
// streaming buffer for one generation, survives HTTP disconnect. the producer appends SSE bytes,
// readers drain from any offset via read_from. keyed by conversation_id, one conv = one live session
// streaming buffer for one generation, survives HTTP disconnect. the producer appends raw SSE
// bytes, readers drain from any offset via read_from and block until more bytes or finalize.
// keyed by conversation_id: one conv = at most one live session
struct stream_session {
std::string conversation_id;
int64_t started_ts; // unix seconds at construction, used by /v1/streams listing
stream_session(std::string conversation_id_, size_t max_bytes_);
stream_session(const stream_session &) = delete;
stream_session & operator=(const stream_session &) = delete;
// append raw bytes, drops from the front if the cap is reached.
// returns false if the session is already finalized
bool append(const char * data, size_t len);
// mark the session as complete, wakes all pending readers
void finalize();
// drain bytes from offset, calling sink for each chunk. blocks until more
// bytes arrive or finalize is called. returns OK on clean exit, OFFSET_LOST
// if offset falls below the dropped prefix
stream_read_status read_from(size_t offset,
const std::function<bool(const char *, size_t)> & sink,
const std::function<bool()> & should_stop);
bool is_done() const;
bool is_cancelled() const;
size_t total_size() const; // bytes that ever entered the session
size_t dropped_prefix() const; // bytes evicted from the front due to cap
int64_t completed_at() const; // 0 while alive, unix seconds after finalize
// attach the producer stop hook used to cancel its reader, pass an empty function to detach
void set_stop_producer(std::function<void()> fn);
// signal the producer to abort its inference asap via the stop hook, idempotent
void cancel();
private:
mutable std::mutex mu;
std::condition_variable cv;
std::vector<char> buffer;
size_t prefix_dropped;
size_t cap_bytes;
std::atomic<bool> done;
std::atomic<bool> cancelled;
std::atomic<int64_t> completed_ts;
std::function<void()> stop_producer; // protected by mu
};
struct stream_session;
using stream_session_ptr = std::shared_ptr<stream_session>;
// one end of a stream_session pipe. the base holds the session and the shared query, the
// producer and consumer ends derive from it. virtual dtor so each end runs its own teardown:
// base of the producer/consumer pipe ends. virtual dtor so each runs its own teardown:
// the producer finalizes the session, the consumer leaves it untouched
struct stream_pipe {
virtual ~stream_pipe() = default;
// true if the session was cancelled (e.g. via DELETE /v1/stream/<conv_id>)
bool is_cancelled() const;
protected:
@@ -95,7 +37,6 @@ protected:
struct stream_pipe_producer : stream_pipe {
~stream_pipe_producer() override;
// append raw bytes to the session's ring buffer, returns false if already finalized
bool write(const char * data, size_t len);
// mark the natural end on the wire so a later close() is a no-op
@@ -121,83 +62,21 @@ private:
server_http_res * res_ = nullptr;
};
// consumer end: read-only replay of the ring buffer, the destructor does not finalize the session
struct stream_pipe_consumer : stream_pipe {
// drain bytes from offset, calling sink for each available chunk. blocks until more data
// arrives or the session finalizes. should_stop is polled, returns OFFSET_LOST if offset
// fell below the dropped prefix
stream_read_status read(size_t & offset,
const std::function<bool(const char *, size_t)> & sink,
const std::function<bool()> & should_stop);
void server_stream_session_manager_start();
void server_stream_session_manager_stop();
static std::shared_ptr<stream_pipe_consumer> create(stream_session_ptr session);
// route handler factories wired under /v1/stream/* by server.cpp
server_http_context::handler_t server_stream_make_get_handler();
server_http_context::handler_t server_stream_make_lookup_handler();
server_http_context::handler_t server_stream_make_delete_handler();
private:
explicit stream_pipe_consumer(stream_session_ptr session);
};
// extract the X-Conversation-Id header value (case-insensitive), empty when absent
std::string server_stream_conv_id_from_headers(const std::map<std::string, std::string> & headers);
// owns all live sessions, runs a periodic GC to evict expired ones.
// the map is keyed by conversation_id, so the invariant "one conv = at most one
// live session" is enforced at the type level
class stream_session_manager {
public:
stream_session_manager();
~stream_session_manager();
stream_session_manager(const stream_session_manager &) = delete;
stream_session_manager & operator=(const stream_session_manager &) = delete;
// install a new session for this conversation, evicting and cancelling any previous one.
// the conversation_id must be non empty, the caller is responsible for that check.
// returns the new session
stream_session_ptr create_or_replace(const std::string & conversation_id);
// lookup, returns null if unknown or already evicted
stream_session_ptr get(const std::string & conversation_id);
// list every live or recently completed session, used by GET /v1/streams without filter
std::vector<stream_session_ptr> list_all() const;
// remove from the map and finalize, wakes any pending readers
void evict(const std::string & conversation_id);
// signal the producer to cancel asap then evict, used by the explicit user Stop path
void evict_and_cancel(const std::string & conversation_id);
void start_gc();
void stop_gc();
private:
void gc_loop();
mutable std::shared_mutex map_mu;
std::unordered_map<std::string, stream_session_ptr> sessions; // key: conversation_id
std::thread gc_thread;
std::atomic<bool> running;
std::mutex gc_wake_mu;
std::condition_variable gc_wake_cv;
};
// process wide manager, linked by both llama-server and llama-cli. llama-server main() drives
// start_gc/stop_gc, llama-cli leaves it idle. the dtor calls stop_gc() unconditionally so exit
// is safe whether or not the GC thread ran
extern stream_session_manager g_stream_sessions;
// route handler factories operating on g_stream_sessions, wired under /v1/stream/* by server.cpp.
// keeps the resumable stream surface confined to server-stream
server_http_context::handler_t make_stream_get_handler();
server_http_context::handler_t make_streams_lookup_handler();
server_http_context::handler_t make_stream_delete_handler();
// extract the X-Conversation-Id header value (case-insensitive), empty when absent. exposed so
// the router can track which child serves a forwarded POST
std::string stream_conv_id_from_headers(const std::map<std::string, std::string> & headers);
// on an X-Conversation-Id header, create or replace the session and attach a producer pipe to
// res. no-op when absent, called from the server_res_generator constructor
void stream_session_attach_pipe(server_http_res & res, const std::map<std::string, std::string> & headers);
// on an X-Conversation-Id header, create or replace the session and attach a producer pipe to res
void server_stream_session_attach_pipe(server_http_res & res, const std::map<std::string, std::string> & headers);
// should_stop closure that ignores peer disconnect when a pipe is attached, so only an explicit
// DELETE stops the producer and generation keeps flowing into the ring buffer. without a pipe it
// delegates to fallback, the legacy non-resumable flow
std::function<bool()> stream_aware_should_stop(server_http_res * res, std::function<bool()> fallback);
std::function<bool()> server_stream_aware_should_stop(server_http_res * res, std::function<bool()> fallback);
+61 -13
View File
@@ -730,6 +730,10 @@ json server_task_result_cmpl_final::to_json_oaicompat_resp_stream() {
}}
});
if (timings.prompt_n >= 0) {
server_sent_events.back().at("data").push_back({"timings", timings.to_json()});
}
return server_sent_events;
}
@@ -1016,6 +1020,7 @@ void server_task_result_cmpl_partial::update(task_result_state & state) {
thinking_block_started = state.thinking_block_started;
text_block_started = state.text_block_started;
oai_resp_created = state.oai_resp_created;
oai_resp_id = state.oai_resp_id;
oai_resp_reasoning_id = state.oai_resp_reasoning_id;
oai_resp_message_id = state.oai_resp_message_id;
@@ -1024,6 +1029,10 @@ void server_task_result_cmpl_partial::update(task_result_state & state) {
// track if the accumulated message has any reasoning content
anthropic_has_reasoning = !state.chat_msg.reasoning_content.empty();
if (res_type == TASK_RESPONSE_TYPE_OAI_RESP && !state.oai_resp_created && (is_progress || n_decoded == 1)) {
state.oai_resp_created = true;
}
// Pre-compute state updates based on diffs (for next chunk)
for (const common_chat_msg_diff & diff : oaicompat_msg_diffs) {
if (!diff.reasoning_content_delta.empty() && !state.thinking_block_started) {
@@ -1181,7 +1190,7 @@ json server_task_result_cmpl_partial::to_json_oaicompat_chat() {
json server_task_result_cmpl_partial::to_json_oaicompat_resp() {
std::vector<json> events;
if (n_decoded == 1) {
if (!oai_resp_created) {
events.push_back(json {
{"event", "response.created"},
{"data", json {
@@ -1204,6 +1213,18 @@ json server_task_result_cmpl_partial::to_json_oaicompat_resp() {
}},
}},
});
} else if (is_progress) {
events.push_back(json {
{"event", "response.in_progress"},
{"data", json {
{"type", "response.in_progress"},
{"response", json {
{"id", oai_resp_id},
{"object", "response"},
{"status", "in_progress"},
}},
}},
});
}
for (const common_chat_msg_diff & diff : oaicompat_msg_diffs) {
@@ -1302,6 +1323,17 @@ json server_task_result_cmpl_partial::to_json_oaicompat_resp() {
});
}
}
if (!events.empty()) {
json & data = events.back().at("data");
if (timings.prompt_n >= 0) {
data.push_back({"timings", timings.to_json()});
}
if (is_progress) {
data.push_back({"prompt_progress", progress.to_json()});
}
}
return events;
}
@@ -1631,7 +1663,22 @@ server_prompt * server_prompt_cache::alloc(const server_prompt & prompt, size_t
}
}
// next, remove any cached prompts that are fully contained in the current prompt
// calculate checkpoints size to see if it will fit with the prompt
size_t checkpoints_size = 0;
for (const auto & ckpt : prompt.checkpoints) {
checkpoints_size += ckpt.size();
}
const size_t state_size_new = state_size_tgt + state_size_dft + checkpoints_size;
// skip over-limit entries to avoid disturbing the cache
if (limit_size > 0 && state_size_new > limit_size) {
SRV_WRN(" - prompt state size %.3f MiB exceeds cache size limit %.3f MiB, skipping\n",
state_size_new / (1024.0 * 1024.0), limit_size / (1024.0 * 1024.0));
return nullptr;
}
// remove any cached prompts that are fully contained in the current prompt
for (auto it = states.begin(); it != states.end();) {
const int len = it->tokens.get_common_prefix(prompt.tokens);
@@ -1644,6 +1691,16 @@ server_prompt * server_prompt_cache::alloc(const server_prompt & prompt, size_t
}
}
if (limit_size > 0) {
// make room before allocating the new vectors to avoid breaching the limit
while (!states.empty() && size() + state_size_new > limit_size) {
SRV_WRN(" - making room for prompt cache entry, removing oldest entry (size = %.3f MiB)\n",
states.front().size() / (1024.0 * 1024.0));
states.pop_front();
}
}
std::vector<uint8_t> state_data_tgt;
std::vector<uint8_t> state_data_dft;
@@ -1752,12 +1809,7 @@ bool server_prompt_cache::load(server_prompt & prompt, const server_tokens & tok
void server_prompt_cache::update() {
if (limit_size > 0) {
// always keep at least one state, regardless of the limits
while (states.size() > 1 && size() > limit_size) {
if (states.empty()) {
break;
}
while (!states.empty() && size() > limit_size) {
SRV_WRN(" - cache size limit reached, removing oldest entry (size = %.3f MiB)\n", states.front().size() / (1024.0 * 1024.0));
states.pop_front();
@@ -1771,11 +1823,7 @@ void server_prompt_cache::update() {
const size_t limit_tokens_cur = limit_size > 0 ? std::max<size_t>(limit_tokens, limit_size/size_per_token) : limit_tokens;
if (limit_tokens > 0) {
while (states.size() > 1 && n_tokens() > limit_tokens_cur) {
if (states.empty()) {
break;
}
while (!states.empty() && n_tokens() > limit_tokens_cur) {
SRV_WRN(" - cache token limit (%zu, est: %zu) reached, removing oldest entry (size = %.3f MiB)\n",
limit_tokens, limit_tokens_cur, states.front().size() / (1024.0 * 1024.0));
+2
View File
@@ -117,6 +117,7 @@ struct task_result_state {
bool text_block_started = false;
// for OpenAI Responses streaming API
bool oai_resp_created = false;
const std::string oai_resp_id;
const std::string oai_resp_reasoning_id;
const std::string oai_resp_message_id;
@@ -440,6 +441,7 @@ struct server_task_result_cmpl_partial : server_task_result {
bool text_block_started = false;
// for OpenAI Responses API
bool oai_resp_created = false;
std::string oai_resp_id;
std::string oai_resp_reasoning_id;
std::string oai_resp_message_id;
+8 -8
View File
@@ -85,7 +85,7 @@ int llama_server(int argc, char ** argv) {
// start the stream session manager GC right after common init, before any HTTP route can
// touch it. lifecycle is symmetric, stop_gc() runs in clean_up() before backend free
g_stream_sessions.start_gc();
server_stream_session_manager_start();
if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_SERVER)) {
return 1;
@@ -245,8 +245,8 @@ int llama_server(int argc, char ** argv) {
ctx_http.post("/slots/:id_slot", ex_wrapper(routes.post_slots));
// resumable streaming, the conversation_id is the session identity end to end. router and
// child wire different handlers under the same paths: a child binds the local g_stream_sessions
// backed factories, the router binds proxies that resolve the owning child through the
// child wire different handlers under the same paths: a child binds the local session
// factories, the router binds proxies that resolve the owning child through the
// conv_id -> model map
server_http_context::handler_t stream_get_h;
server_http_context::handler_t streams_lookup_h;
@@ -256,9 +256,9 @@ int llama_server(int argc, char ** argv) {
streams_lookup_h = models_routes->router_streams_lookup;
stream_delete_h = models_routes->router_stream_delete;
} else {
stream_get_h = make_stream_get_handler();
streams_lookup_h = make_streams_lookup_handler();
stream_delete_h = make_stream_delete_handler();
stream_get_h = server_stream_make_get_handler();
streams_lookup_h = server_stream_make_lookup_handler();
stream_delete_h = server_stream_make_delete_handler();
}
ctx_http.get ("/v1/stream/:conv_id", ex_wrapper(stream_get_h));
// POST /v1/streams/lookup with body {"conversation_ids": [...]}. you can only ask for ids
@@ -343,7 +343,7 @@ int llama_server(int argc, char ** argv) {
clean_up = [&models_routes]() {
SRV_INF("%s: cleaning up before exit...\n", __func__);
// stop the session GC first, it finalizes live sessions and wakes pending readers
g_stream_sessions.stop_gc();
server_stream_session_manager_stop();
if (models_routes.has_value()) {
models_routes->stopping.store(true); // maybe redundant, but just to be safe
models_routes->models.unload_all();
@@ -371,7 +371,7 @@ int llama_server(int argc, char ** argv) {
clean_up = [&ctx_http, &ctx_server]() {
SRV_INF("%s: cleaning up before exit...\n", __func__);
// stop the session GC first, it finalizes live sessions and wakes pending readers
g_stream_sessions.stop_gc();
server_stream_session_manager_stop();
ctx_http.stop();
ctx_server.terminate();
llama_backend_free();
@@ -71,3 +71,44 @@ def test_responses_stream_with_openai_library():
assert r.response.output[0].id.startswith("msg_")
assert gathered_text == r.response.output_text
assert match_regex("(Suddenly)+", r.response.output_text)
def test_responses_stream_with_llama_telemetry():
global server
server.n_ctx = 256
server.n_batch = 32
server.n_slots = 1
server.start()
saw_progress = False
saw_delta_timings = False
completed = None
res = server.make_stream_request("POST", "/responses", data={
"input": "This is a test" * 10,
"max_output_tokens": 8,
"temperature": 0.8,
"stream": True,
"timings_per_token": True,
"return_progress": True,
})
for data in res:
if "prompt_progress" in data:
assert data["type"] == "response.in_progress"
assert data["prompt_progress"]["total"] > 0
assert data["prompt_progress"]["processed"] >= data["prompt_progress"]["cache"]
saw_progress = True
if "timings" in data:
assert "prompt_per_second" in data["timings"]
assert "predicted_per_second" in data["timings"]
if data["type"] == "response.output_text.delta":
saw_delta_timings = True
if data["type"] == "response.completed":
completed = data
assert saw_progress
assert saw_delta_timings
assert completed is not None
assert "usage" in completed["response"]
assert "timings" in completed
+4 -4
View File
@@ -11,7 +11,7 @@
"@chromatic-com/storybook": "5.0.0",
"@eslint/compat": "1.4.1",
"@eslint/js": "9.39.2",
"@internationalized/date": "3.10.1",
"@internationalized/date": "3.12.2",
"@lucide/svelte": "0.515.0",
"@modelcontextprotocol/sdk": "1.26.0",
"@playwright/test": "1.56.1",
@@ -2981,9 +2981,9 @@
}
},
"node_modules/@internationalized/date": {
"version": "3.10.1",
"resolved": "https://registry.npmjs.org/@internationalized/date/-/date-3.10.1.tgz",
"integrity": "sha512-oJrXtQiAXLvT9clCf1K4kxp3eKsQhIaZqxEyowkBcsvZDdZkbWrVmnGknxs5flTD0VGsxrxKgBCZty1EzoiMzA==",
"version": "3.12.2",
"resolved": "https://registry.npmjs.org/@internationalized/date/-/date-3.12.2.tgz",
"integrity": "sha512-FY1Y+H64NDs+HAF6omlnWxm3mEpfgaCSWtL5l551ZZfImA+kGjPFgrnJrGjH6lfmLL0g8Z/mBu1R3kufeCp6Jw==",
"dev": true,
"license": "Apache-2.0",
"dependencies": {
+1 -1
View File
@@ -30,7 +30,7 @@
"@chromatic-com/storybook": "5.0.0",
"@eslint/compat": "1.4.1",
"@eslint/js": "9.39.2",
"@internationalized/date": "3.10.1",
"@internationalized/date": "3.12.2",
"@lucide/svelte": "0.515.0",
"@modelcontextprotocol/sdk": "1.26.0",
"@playwright/test": "1.56.1",
@@ -11,7 +11,8 @@
} from '$lib/constants';
import {
ChatFormActionAddToolsSubmenu,
ChatFormActionAddMcpServersSubmenu
ChatFormActionAddMcpServersSubmenu,
ChatFormActionAddReasoningSubmenu
} from '$lib/components/app';
import { useAttachmentMenu } from '$lib/hooks/use-attachment-menu.svelte';
@@ -92,7 +93,11 @@
</Tooltip.Content>
</Tooltip.Root>
<DropdownMenu.Content align="start" class="w-48">
<DropdownMenu.Content align="start" class="w-52">
<ChatFormActionAddReasoningSubmenu />
<DropdownMenu.Separator />
<DropdownMenu.Sub>
<DropdownMenu.SubTrigger class="flex cursor-pointer items-center gap-2">
<File class="h-4 w-4" />
@@ -2,7 +2,7 @@
import { Lightbulb, LightbulbOff, Check, Info } from '@lucide/svelte';
import * as DropdownMenu from '$lib/components/ui/dropdown-menu';
import * as Tooltip from '$lib/components/ui/tooltip';
import { ReasoningEffort, MessageRole } from '$lib/enums';
import { ReasoningEffort } from '$lib/enums';
import { REASONING_EFFORT_TOKENS } from '$lib/constants/reasoning-effort-tokens';
import { REASONING_EFFORT_LEVELS } from '$lib/constants/reasoning-effort';
import type { ReasoningEffortLevel } from '$lib/types';
@@ -18,31 +18,23 @@
import { isRouterMode } from '$lib/stores/server.svelte';
import type { DatabaseMessage } from '$lib/types/database';
let thinkingEnabled = $derived(conversationsStore.getThinkingEnabled());
let currentEffort = $derived(conversationsStore.getReasoningEffort());
let isOff = $derived(!thinkingEnabled);
let tooltipText = $derived(thinkingEnabled ? `${currentEffort} Reasoning` : 'Disabled Reasoning');
let subOpen = $state(false);
// Get conversation model from message history
let conversationModel = $derived(
chatStore.getConversationModel(activeMessages() as DatabaseMessage[])
);
// Fallback: if model props aren't available, check if any assistant messages
// for this model in the active conversation have reasoning content.
let modelSupportsThinkingFromMessages = $derived.by(() => {
const modelId = isRouterMode() ? modelsStore.selectedModelName || conversationModel : null;
if (!modelId) return false;
const messages = conversationsStore.activeMessages;
return messages.some(
(m: DatabaseMessage) =>
m.role === MessageRole.ASSISTANT && m.model === modelId && !!m.reasoningContent
(m) => m.role === 'assistant' && m.model === modelId && !!m.reasoningContent
);
});
// Check if model supports thinking. Primary: chat template from /props.
// Fallback: message history (reasoning content in assistant messages).
let modelSupportsThinking = $derived.by(() => {
loadedModelIds();
propsCacheVersion();
@@ -52,15 +44,15 @@
return checkModelSupportsThinking(modelId ?? '') || modelSupportsThinkingFromMessages;
}
// In non-router mode, use the built-in supportsThinking
return supportsThinking() || modelSupportsThinkingFromMessages;
});
// Check if current item is selected
let thinkingEnabled = $derived(conversationsStore.getThinkingEnabled());
let currentEffort = $derived(conversationsStore.getReasoningEffort());
let isOff = $derived(!thinkingEnabled);
function isSelected(item: ReasoningEffortLevel): boolean {
if (item.isOff) {
return isOff;
}
if (item.isOff) return isOff;
return thinkingEnabled && currentEffort === item.value;
}
@@ -76,39 +68,30 @@
</script>
{#if modelSupportsThinking}
<DropdownMenu.Root bind:open={subOpen}>
<Tooltip.Root>
<Tooltip.Trigger>
<DropdownMenu.Trigger
class={[
'flex h-6 w-6 cursor-pointer items-center justify-center rounded-full p-0 transition-colors focus:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2',
thinkingEnabled ? 'bg-amber-400/10 hover:bg-amber-400/20' : 'bg-muted'
]}
aria-label={`${tooltipText}. Click to configure.`}
>
{#if thinkingEnabled}
<Lightbulb class="h-3 w-3 text-amber-400" />
{:else}
<LightbulbOff class="h-3 w-3 text-muted-foreground" />
{/if}
</DropdownMenu.Trigger>
</Tooltip.Trigger>
<DropdownMenu.Sub bind:open={subOpen}>
<DropdownMenu.SubTrigger class="flex cursor-pointer items-center gap-2">
{#if thinkingEnabled}
<Lightbulb class="h-4 w-4 shrink-0 text-amber-400" />
{:else}
<LightbulbOff class="h-4 w-4 shrink-0 text-muted-foreground" />
{/if}
<Tooltip.Content>
<p class="capitalize">{tooltipText}</p>
</Tooltip.Content>
</Tooltip.Root>
<span class="text-sm inline-flex gap-2 {!thinkingEnabled ? 'text-muted-foreground' : ''}">
Reasoning
<DropdownMenu.Content
align="start"
class="w-60 rounded-xl bg-popover p-3 text-popover-foreground shadow-md outline-none"
<span class="capitalize text-muted-foreground">
{thinkingEnabled ? currentEffort : 'off'}
</span>
</span>
</DropdownMenu.SubTrigger>
<DropdownMenu.SubContent
class="w-60 bg-popover p-1.5 text-popover-foreground shadow-md outline-none"
>
<div class="mb-2 px-2.5 text-sm font-medium">Reasoning effort</div>
{#each REASONING_EFFORT_LEVELS as level (level.value)}
<button
type="button"
class="flex w-full cursor-pointer items-center gap-2 rounded-lg px-2.5 py-2 text-left text-sm transition-colors hover:bg-accent"
class="flex w-full cursor-pointer items-center gap-3 rounded-md px-2 py-1.75 text-left text-sm transition-colors hover:bg-accent"
class:bg-accent={isSelected(level)}
onclick={() => handleSelection(level)}
>
@@ -140,6 +123,6 @@
{/if}
</button>
{/each}
</DropdownMenu.Content>
</DropdownMenu.Root>
</DropdownMenu.SubContent>
</DropdownMenu.Sub>
{/if}
@@ -7,14 +7,20 @@
ChatFormActionModels,
ChatFormActionRecord,
ChatFormActionSubmit,
ChatFormReasoningToggle
ChatFormContextGauge
} from '$lib/components/app';
import { FileTypeCategory } from '$lib/enums';
import { FileTypeCategory, MessageRole } from '$lib/enums';
import { mcpStore } from '$lib/stores/mcp.svelte';
import { config } from '$lib/stores/settings.svelte';
import { conversationsStore } from '$lib/stores/conversations.svelte';
import { activeMessages, conversationsStore } from '$lib/stores/conversations.svelte';
import {
activeProcessingState,
isChatStreaming,
isLoading as chatIsLoading
} from '$lib/stores/chat.svelte';
import { getFileTypeCategory } from '$lib/utils';
import { goto } from '$app/navigation';
import { page } from '$app/state';
import { ROUTES } from '$lib/constants/routes';
interface Props {
@@ -93,6 +99,36 @@
let activeMessage = $derived(
conversationsStore.activeMessages[conversationsStore.activeMessages.length - 1]
);
let hasProcessedTokens = $derived.by(() => {
if (!page.params.id) return false;
const messages = activeMessages() as DatabaseMessage[];
let totalHistoricalTokens = 0;
for (const m of messages) {
if (m.role !== MessageRole.ASSISTANT) continue;
const timings = m.timings;
if (!timings) continue;
const agenticLlm = timings.agentic?.llm;
if (agenticLlm?.prompt_n != null || agenticLlm?.predicted_n != null) {
totalHistoricalTokens += (agenticLlm?.prompt_n ?? 0) + (agenticLlm?.predicted_n ?? 0);
} else {
totalHistoricalTokens += (timings.prompt_n ?? 0) + (timings.predicted_n ?? 0);
}
}
if (totalHistoricalTokens > 0) return true;
if (!chatIsLoading() && !isChatStreaming()) return false;
const processingState = activeProcessingState();
if (!processingState) return false;
const livePromptTokens = Math.max(
processingState.promptTokens ?? 0,
processingState.promptProgress?.processed ?? 0
);
const liveOutputTokens = processingState.outputTokensUsed ?? 0;
return livePromptTokens > 0 || liveOutputTokens > 0;
});
</script>
<div
@@ -100,7 +136,7 @@
style="container-type: inline-size"
>
{#if showAddButton}
<div class="mr-auto flex items-center gap-3">
<div class="mr-auto flex items-center gap-2">
<ChatFormActionsAdd
{disabled}
{hasAudioModality}
@@ -117,8 +153,10 @@
</div>
{/if}
<div class="flex items-center gap-2">
<ChatFormReasoningToggle />
<div class="flex items-center gap-1.5">
{#if hasProcessedTokens}
<ChatFormContextGauge />
{/if}
{#if showModelSelector}
<ChatFormActionModels
@@ -6,6 +6,7 @@
import { REASONING_EFFORT_TOKENS } from '$lib/constants/reasoning-effort-tokens';
import { REASONING_EFFORT_LEVELS } from '$lib/constants/reasoning-effort';
import type { ReasoningEffortLevel } from '$lib/types';
import { DIALOG_SUBMENU_CONTENT } from '$lib/constants/css-classes';
import {
modelsStore,
checkModelSupportsThinking,
@@ -71,9 +72,7 @@
{#if modelSupportsThinking}
<DropdownMenu.Sub bind:open={subOpen}>
<DropdownMenu.SubTrigger
class="flex cursor-pointer items-center gap-2 rounded-md px-2.5 py-1.5 text-sm transition-colors outline-none hover:bg-accent focus:bg-accent"
>
<DropdownMenu.SubTrigger class="flex cursor-pointer items-center gap-2">
{#if thinkingEnabled}
<Lightbulb class="h-4 w-4 shrink-0 text-amber-400" />
{:else}
@@ -89,23 +88,15 @@
{/if}
</DropdownMenu.SubTrigger>
<DropdownMenu.SubContent
class="w-60 rounded-xl bg-popover p-3 text-popover-foreground shadow-md outline-none data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 data-[state=closed]:animate-out data-[state=closed]:fade-out-0 data-[state=closed]:zoom-out-95 data-[state=open]:animate-in data-[state=open]:fade-in-0 data-[state=open]:zoom-in-95"
>
<DropdownMenu.SubContent class={DIALOG_SUBMENU_CONTENT}>
{#each REASONING_EFFORT_LEVELS as level (level.value)}
<button
type="button"
class="flex w-full cursor-pointer items-center gap-2 rounded-lg px-2.5 py-2 text-left text-sm transition-colors hover:bg-accent"
class="flex w-full cursor-pointer items-center gap-2"
class:bg-accent={isSelected(level)}
onclick={() => handleSelection(level)}
>
{#if isSelected(level)}
<Check class="h-4 w-4 shrink-0 text-foreground" />
{:else}
<div class="h-4 w-4 shrink-0"></div>
{/if}
<span class="flex-1">{level.label}</span>
<span class="flex-1 text-left">{level.label}</span>
{#if !level.isOff}
<span class="text-[11px] text-muted-foreground opacity-60">
@@ -125,6 +116,10 @@
</Tooltip.Content>
</Tooltip.Root>
{/if}
{#if isSelected(level)}
<Check class="h-4 w-4 shrink-0 text-foreground" />
{/if}
</button>
{/each}
</DropdownMenu.SubContent>
@@ -0,0 +1,108 @@
<script lang="ts">
import { untrack } from 'svelte';
import * as HoverCard from '$lib/components/ui/hover-card';
import { activeConversation, activeMessages } from '$lib/stores/conversations.svelte';
import { chatStore, isChatStreaming, isLoading } from '$lib/stores/chat.svelte';
import { formatParameters } from '$lib/utils/formatters';
import { useContextGauge } from '$lib/hooks/use-context-gauge.svelte';
import ContextGaugeDial from './ContextGaugeDial.svelte';
import ContextGaugeDetails from './ContextGaugeDetails.svelte';
import ContextGaugeLoadModel from './ContextGaugeLoadModel.svelte';
import { colorLevelBgClass, colorLevelTextClass } from './context-gauge';
const gauge = useContextGauge();
$effect(() => {
const conv = activeConversation();
untrack(() => chatStore.setActiveProcessingConversation(conv?.id ?? null));
});
$effect(() => {
const conv = activeConversation();
const messages = activeMessages() as DatabaseMessage[];
if (!conv) return;
if (isLoading() || isChatStreaming()) return;
if (messages.length === 0) {
untrack(() => chatStore.clearProcessingState(conv.id));
return;
}
untrack(() => chatStore.restoreProcessingStateFromMessages(messages, conv.id));
});
$effect(() => {
gauge.startMonitoring();
});
const showProgressBar = $derived(
gauge.contextTotal !== null &&
gauge.contextTotal > 0 &&
(gauge.activeModelId !== null || gauge.isActiveModelLoaded)
);
</script>
<HoverCard.Root>
<HoverCard.Trigger class="flex h-5 w-5 cursor-default items-center justify-center">
<ContextGaugeDial percent={gauge.contextPercent} level={gauge.colorLevel} />
</HoverCard.Trigger>
<HoverCard.Content
side="bottom"
class="z-50 w-64 rounded-lg border border-border/50 bg-popover p-3 text-popover-foreground shadow-lg"
>
<div class="flex flex-col gap-2">
<div class="flex items-center gap-2">
<span class="font-medium">Context</span>
<span class="text-muted-foreground">·</span>
<span class="font-mono text-muted-foreground">
{formatParameters(gauge.contextUsed)}
/ {gauge.contextTotal !== null ? formatParameters(gauge.contextTotal) : '-'}
</span>
</div>
{#if gauge.activeModelId !== null && !gauge.isActiveModelLoaded}
<ContextGaugeLoadModel
modelId={gauge.activeModelId}
isLoading={gauge.isActiveModelLoading}
onLoad={gauge.loadModel}
/>
{:else if showProgressBar}
<div class="h-1.5 w-full overflow-hidden rounded-full bg-muted">
<div
class="h-full rounded-full transition-all duration-300 {colorLevelBgClass(
gauge.colorLevel
)}"
style="width: {gauge.contextPercent}%"
></div>
</div>
<div class="flex justify-between text-xs text-muted-foreground">
<span>
<span class={colorLevelTextClass(gauge.colorLevel)}>{gauge.contextPercent}%</span> used
</span>
<span>
{formatParameters((gauge.contextTotal ?? 0) - gauge.contextUsed)} remaining
</span>
</div>
{:else}
<div class="text-xs text-muted-foreground">No context info available</div>
{/if}
{#if gauge.hasAnyUsage}
<ContextGaugeDetails
currentRead={gauge.currentRead}
currentFresh={gauge.currentFresh}
currentCache={gauge.currentCache}
currentOutput={gauge.currentOutput}
kvTotal={gauge.kvTotal}
cumulativeRead={gauge.cumulativeRead}
cumulativeOutput={gauge.cumulativeOutput}
cumulativeCacheTotal={gauge.cumulativeCacheTotal}
averageTokensPerSecond={gauge.averageTokensPerSecond}
transientDetails={gauge.transientDetails}
/>
{/if}
</div>
</HoverCard.Content>
</HoverCard.Root>
@@ -0,0 +1,20 @@
<script lang="ts">
interface Props {
label: string;
value: string;
subtitle?: string;
}
let { label, value, subtitle }: Props = $props();
</script>
<div class="grid gap-1.5">
<div class="flex items-baseline justify-between">
<span class="text-muted-foreground">{label}</span>
<span class="font-mono text-muted-foreground">{value}</span>
</div>
{#if subtitle}
<div class="text-[10px] leading-tight text-muted-foreground/70">{subtitle}</div>
{/if}
</div>
@@ -0,0 +1,122 @@
<script lang="ts">
import { ChevronDown } from '@lucide/svelte';
import * as Collapsible from '$lib/components/ui/collapsible';
import { STATS_UNITS } from '$lib/constants';
import ContextGaugeDetailRow from './ContextGaugeDetailRow.svelte';
interface Props {
currentRead: number;
currentFresh: number;
currentCache: number;
currentOutput: number;
kvTotal: number;
cumulativeRead: number;
cumulativeOutput: number;
cumulativeCacheTotal: number;
averageTokensPerSecond: number | null;
transientDetails: string[];
}
let {
currentRead,
currentFresh,
currentCache,
currentOutput,
kvTotal,
cumulativeRead,
cumulativeOutput,
cumulativeCacheTotal,
averageTokensPerSecond,
transientDetails
}: Props = $props();
let open = $state(false);
const hasCumulative = $derived(cumulativeRead > 0 || cumulativeOutput > 0);
const hasCurrent = $derived(currentRead > 0 || currentOutput > 0);
</script>
<Collapsible.Root bind:open class="mt-3 border-t border-border/50 pt-4">
<Collapsible.Trigger
class="flex w-full cursor-pointer items-center gap-1 text-xs text-muted-foreground hover:text-foreground"
>
<span>Token usage details</span>
<ChevronDown class={'ml-auto h-3 w-3 transition-transform' + (open ? ' rotate-180' : '')} />
</Collapsible.Trigger>
<Collapsible.Content class="flex flex-col gap-4 text-xs pt-4">
{#if hasCumulative}
<div>
<h3 class="text-[11px] font-medium uppercase tracking-wide text-muted-foreground/70 mb-2">
Across all turns
</h3>
<div class="flex flex-col gap-2">
{#if cumulativeRead > 0}
<ContextGaugeDetailRow
label="Prompt tokens evaluated"
value={`${cumulativeRead.toLocaleString()} tok`}
subtitle={cumulativeCacheTotal > 0
? `${cumulativeCacheTotal.toLocaleString()} reused from KV cache`
: undefined}
/>
{/if}
{#if cumulativeOutput > 0}
<ContextGaugeDetailRow
label="Tokens generated"
value={`${cumulativeOutput.toLocaleString()} tok`}
/>
{/if}
</div>
</div>
{/if}
{#if hasCurrent}
<div>
<h3 class="text-[11px] font-medium uppercase tracking-wide text-muted-foreground/70 mb-2">
This turn · KV cache
</h3>
<div class="flex flex-col gap-2">
{#if currentRead > 0}
<ContextGaugeDetailRow
label="Prompt"
value={`${currentRead.toLocaleString()} tok`}
subtitle={currentCache > 0
? `${currentFresh.toLocaleString()} fresh + ${currentCache.toLocaleString()} cached`
: undefined}
/>
{/if}
{#if currentOutput > 0}
<ContextGaugeDetailRow
label="Generated"
value={`${currentOutput.toLocaleString()} tok`}
/>
{/if}
<div class="pt-1 mt-0.5 border-t border-border/30">
<div class="flex justify-between">
<span class="text-muted-foreground">KV cache total</span>
<span class="font-mono font-medium">{kvTotal.toLocaleString()} tok</span>
</div>
</div>
</div>
</div>
{/if}
{#if averageTokensPerSecond !== null}
<div class="pt-1.5 mt-1 border-t border-border/30">
<ContextGaugeDetailRow
label="Avg speed"
value={`${averageTokensPerSecond.toFixed(1)}${STATS_UNITS.TOKENS_PER_SECOND}`}
/>
</div>
{/if}
{#each transientDetails as detail (detail)}
<div class="font-mono text-muted-foreground">{detail}</div>
{/each}
</Collapsible.Content>
</Collapsible.Root>
@@ -0,0 +1,43 @@
<script lang="ts">
import type { ColorLevel } from './context-gauge';
import { colorLevelTextClass } from './context-gauge';
interface Props {
percent: number | null;
level: ColorLevel;
size?: 'sm' | 'md';
}
let { percent, level, size = 'sm' }: Props = $props();
const RADIUS = 11;
const CIRCUMFERENCE = 2 * Math.PI * RADIUS;
const strokeLevelClass = $derived(colorLevelTextClass(level));
const dimensions = $derived(size === 'md' ? 'h-6 w-6' : 'h-5 w-5');
const strokeWidth = $derived(size === 'md' ? 4 : 3);
</script>
<svg viewBox="0 0 32 32" fill="none" class={dimensions}>
<circle
cx="16"
cy="16"
r={RADIUS}
stroke="currentColor"
stroke-opacity="0.1"
stroke-width={strokeWidth}
/>
<circle
cx="16"
cy="16"
r={RADIUS}
class="transition-colors duration-300 {strokeLevelClass}"
stroke="currentColor"
stroke-width={strokeWidth}
stroke-linecap="round"
stroke-dasharray={CIRCUMFERENCE}
stroke-dashoffset={percent !== null ? CIRCUMFERENCE * (1 - percent / 100) : CIRCUMFERENCE}
transform="rotate(-90 16 16)"
/>
</svg>
@@ -0,0 +1,24 @@
<script lang="ts">
import { Loader2 } from '@lucide/svelte';
import { Button } from '$lib/components/ui/button';
interface Props {
modelId: string | null;
isLoading: boolean;
onLoad: () => void;
}
let { modelId, isLoading, onLoad }: Props = $props();
</script>
{#if modelId !== null && !isLoading}
<div class="flex flex-col gap-2 border-t border-border/50 pt-2 text-xs text-muted-foreground">
<span>Available context size is only visible once the model is loaded.</span>
<Button size="sm" variant="secondary" class="self-start" onclick={onLoad}>Load model</Button>
</div>
{:else if isLoading}
<div class="flex items-center gap-2 border-t border-border/50 pt-2 text-xs text-muted-foreground">
<Loader2 class="h-3.5 w-3.5 animate-spin" />
<span>Loading model...</span>
</div>
{/if}
@@ -0,0 +1,37 @@
export type ColorLevel = 'ok' | 'warning' | 'critical' | 'neutral';
const WARNING_THRESHOLD = 80;
const CRITICAL_THRESHOLD = 95;
export function colorLevelFromPercent(percent: number | null): ColorLevel {
if (percent === null) return 'neutral';
if (percent >= CRITICAL_THRESHOLD) return 'critical';
if (percent >= WARNING_THRESHOLD) return 'warning';
return 'ok';
}
export function colorLevelTextClass(level: ColorLevel): string {
switch (level) {
case 'critical':
return 'text-red-400';
case 'warning':
return 'text-amber-400';
case 'ok':
return 'text-muted-foreground';
default:
return 'text-muted-foreground';
}
}
export function colorLevelBgClass(level: ColorLevel): string {
switch (level) {
case 'critical':
return 'bg-red-500';
case 'warning':
return 'bg-amber-500';
case 'ok':
return 'bg-green-500';
default:
return 'bg-muted';
}
}
@@ -24,6 +24,8 @@
message: DatabaseMessage;
toolMessages?: DatabaseMessage[];
isLastAssistantMessage?: boolean;
isLastUserMessage?: boolean;
nextAssistantMessage?: DatabaseMessage | null;
siblingInfo?: ChatMessageSiblingInfo | null;
}
@@ -32,6 +34,8 @@
message,
toolMessages = [],
isLastAssistantMessage = false,
isLastUserMessage = false,
nextAssistantMessage = null,
siblingInfo = null
}: Props = $props();
@@ -359,7 +363,9 @@
<ChatMessageUser
class={className}
{deletionInfo}
{isLastUserMessage}
{message}
{nextAssistantMessage}
onConfirmDelete={handleConfirmDelete}
onCopy={handleCopy}
onDelete={handleDelete}
@@ -11,11 +11,10 @@
import { useProcessingState } from '$lib/hooks/use-processing-state.svelte';
import { isLoading, isChatStreaming } from '$lib/stores/chat.svelte';
import { copyToClipboard, deriveAgenticSections, modelLoadProgressText } from '$lib/utils';
import { AgenticSectionType } from '$lib/enums';
import { AgenticSectionType, ChatMessageStatisticsMode } from '$lib/enums';
import { REASONING_TAGS } from '$lib/constants/agentic';
import { tick } from 'svelte';
import { fade } from 'svelte/transition';
import { MessageRole, ChatMessageStatsView } from '$lib/enums';
import { MessageRole } from '$lib/enums';
import { config } from '$lib/stores/settings.svelte';
import { isRouterMode } from '$lib/stores/server.svelte';
import { modelsStore } from '$lib/stores/models.svelte';
@@ -122,62 +121,6 @@
return parts.join('\n\n\n');
});
let activeStatsView = $state<ChatMessageStatsView>(ChatMessageStatsView.GENERATION);
let statsContainerEl: HTMLDivElement | undefined = $state();
function getScrollParent(el: HTMLElement): HTMLElement | null {
let parent = el.parentElement;
while (parent) {
const style = getComputedStyle(parent);
if (/(auto|scroll)/.test(style.overflowY)) {
return parent;
}
parent = parent.parentElement;
}
return null;
}
async function handleStatsViewChange(view: ChatMessageStatsView) {
const el = statsContainerEl;
if (!el) {
activeStatsView = view;
return;
}
const scrollParent = getScrollParent(el);
if (!scrollParent) {
activeStatsView = view;
return;
}
const yBefore = el.getBoundingClientRect().top;
activeStatsView = view;
await tick();
const delta = el.getBoundingClientRect().top - yBefore;
if (delta !== 0) {
scrollParent.scrollTop += delta;
}
// Correct any drift after browser paint
requestAnimationFrame(() => {
const drift = el.getBoundingClientRect().top - yBefore;
if (Math.abs(drift) > 1) {
scrollParent.scrollTop += drift;
}
});
}
let highlightAgenticTurns = $derived(
isAgentic &&
(currentConfig.alwaysShowAgenticTurns || activeStatsView === ChatMessageStatsView.SUMMARY)
);
let displayedModel = $derived(message.model ?? null);
// model being switched to while it loads, so the selector bar tracks it
@@ -291,7 +234,6 @@
{toolMessages}
isStreaming={isChatStreaming()}
{isLastAssistantMessage}
highlightTurns={highlightAgenticTurns}
/>
{/if}
{:else}
@@ -315,10 +257,7 @@
<div class="info my-6 grid gap-4 tabular-nums">
{#if displayedModel}
<div
bind:this={statsContainerEl}
class="inline-flex flex-wrap items-start gap-2 text-xs text-muted-foreground"
>
<div class="inline-flex flex-wrap items-start gap-2 text-xs text-muted-foreground">
{#if isRouter}
<ModelsSelectorDropdown
currentModel={pendingModel ?? displayedModel}
@@ -347,28 +286,25 @@
{#if currentConfig.showMessageStats && message.timings && message.timings.predicted_n && message.timings.predicted_ms}
{@const agentic = message.timings.agentic}
<ChatMessageStatistics
mode={ChatMessageStatisticsMode.GENERATION}
promptTokens={agentic ? agentic.llm.prompt_n : message.timings.prompt_n}
promptMs={agentic ? agentic.llm.prompt_ms : message.timings.prompt_ms}
predictedTokens={agentic ? agentic.llm.predicted_n : message.timings.predicted_n}
predictedMs={agentic ? agentic.llm.predicted_ms : message.timings.predicted_ms}
agenticTimings={agentic}
onActiveViewChange={handleStatsViewChange}
/>
{:else if isLoading() && currentConfig.showMessageStats}
{@const liveStats = processingState.getLiveProcessingStats()}
{@const genStats = processingState.getLiveGenerationStats()}
{@const promptProgress = processingState.processingState?.promptProgress}
{@const isStillProcessingPrompt =
promptProgress && promptProgress.processed < promptProgress.total}
{#if liveStats || genStats}
{#if genStats}
<ChatMessageStatistics
mode={ChatMessageStatisticsMode.GENERATION}
isLive
isProcessingPrompt={!!isStillProcessingPrompt}
promptTokens={liveStats?.tokensProcessed}
promptMs={liveStats?.timeMs}
predictedTokens={genStats?.tokensGenerated}
predictedMs={genStats?.timeMs}
predictedTokens={genStats.tokensGenerated}
predictedMs={genStats.timeMs}
/>
{/if}
{/if}
@@ -2,10 +2,14 @@
import {
ChatMessageActionIcons,
ChatMessageEditForm,
ChatMessageStatistics,
ChatMessageUserBubble
} from '$lib/components/app/chat';
import { getMessageEditContext } from '$lib/contexts';
import { MessageRole } from '$lib/enums';
import { useProcessingState } from '$lib/hooks/use-processing-state.svelte';
import { isLoading } from '$lib/stores/chat.svelte';
import { MessageRole, ChatMessageStatisticsMode } from '$lib/enums';
import { config } from '$lib/stores/settings.svelte';
interface Props {
class?: string;
@@ -17,6 +21,8 @@
assistantMessages: number;
messageTypes: string[];
} | null;
isLastUserMessage?: boolean;
nextAssistantMessage?: DatabaseMessage | null;
showDeleteDialog: boolean;
onEdit: () => void;
onDelete: () => void;
@@ -32,6 +38,8 @@
message,
siblingInfo = null,
deletionInfo,
isLastUserMessage = false,
nextAssistantMessage = null,
showDeleteDialog,
onEdit,
onDelete,
@@ -44,6 +52,37 @@
// Get contexts
const editCtx = getMessageEditContext();
const processingState = useProcessingState();
const currentConfig = $derived(config());
const isActivelyProcessing = $derived(isLastUserMessage && isLoading());
// For agentic turns, prefer the cumulative agentic.llm totals over per-call timings.
let storedReadingStats = $derived.by(() => {
const timings = nextAssistantMessage?.timings;
if (!timings?.prompt_n || !timings?.prompt_ms) return null;
const agentic = timings.agentic;
return {
promptTokens: agentic ? agentic.llm.prompt_n : timings.prompt_n,
promptMs: agentic ? agentic.llm.prompt_ms : timings.prompt_ms
};
});
let showStoredReadingStats = $derived(
Boolean(currentConfig.showMessageStats) && storedReadingStats !== null
);
let showLiveReadingStats = $derived(
Boolean(currentConfig.showMessageStats) && isActivelyProcessing && storedReadingStats === null
);
$effect(() => {
if (showLiveReadingStats) {
processingState.startMonitoring();
}
});
</script>
<div
@@ -60,6 +99,37 @@
renderMarkdown={true}
/>
{#if showStoredReadingStats}
<!-- Reading stats sourced from the assistant message that followed this turn -->
<div class="info my-2 grid w-full justify-items-end gap-4 tabular-nums">
<div
class="inline-flex flex-wrap items-start justify-end gap-2 text-xs text-muted-foreground"
>
<ChatMessageStatistics
mode={ChatMessageStatisticsMode.READING}
promptTokens={storedReadingStats!.promptTokens}
promptMs={storedReadingStats!.promptMs}
/>
</div>
</div>
{:else if showLiveReadingStats}
{@const liveStats = processingState.getLiveProcessingStats()}
{#if liveStats}
<div class="info my-2 grid w-full justify-items-end gap-4 tabular-nums">
<div
class="inline-flex flex-wrap items-start justify-end gap-2 text-xs text-muted-foreground"
>
<ChatMessageStatistics
mode={ChatMessageStatisticsMode.READING}
isLive
promptTokens={liveStats.tokensProcessed}
promptMs={liveStats.timeMs}
/>
</div>
</div>
{/if}
{/if}
{#if message.timestamp}
<div class="max-w-[80%]">
<ChatMessageActionIcons
@@ -2,7 +2,6 @@
import { ActionIcon, ChatMessageEditForm, ChatMessageUserBubble } from '$lib/components/app';
import { fadeInView } from '$lib/actions/fade-in-view.svelte';
import { ArrowUp, Edit, Trash2 } from '@lucide/svelte';
import { getProcessingInfoContext } from '$lib/contexts';
import { useMessageEditContext } from '$lib/hooks/use-message-edit-context.svelte';
interface Props {
@@ -23,9 +22,6 @@
onDelete
}: Props = $props();
const processingInfoCtx = getProcessingInfoContext();
let showProcessingInfo = $derived(processingInfoCtx.showProcessingInfo);
const editCtx = useMessageEditContext({
getContent: () => content,
getExtras: () => extras,
@@ -36,9 +32,7 @@
<div
use:fadeInView
aria-label="Pending user message"
class="group flex flex-col items-end gap-3 transition-opacity hover:opacity-80 md:gap-2 {className} sticky {showProcessingInfo
? 'bottom-44'
: 'bottom-32'}"
class="group flex flex-col items-end gap-3 transition-opacity hover:opacity-80 md:gap-2 {className} sticky bottom-32"
role="group"
>
{#if editCtx.isEditing}
@@ -41,15 +41,13 @@
toolMessages?: DatabaseMessage[];
isStreaming?: boolean;
isLastAssistantMessage?: boolean;
highlightTurns?: boolean;
}
let {
message,
toolMessages = [],
isStreaming = false,
isLastAssistantMessage = false,
highlightTurns = false
isLastAssistantMessage = false
}: Props = $props();
let expandedStates: Record<number, boolean> = $state({});
@@ -57,6 +55,7 @@
const showToolCallInProgress = $derived(config().showToolCallInProgress as boolean);
const showThoughtInProgress = $derived(config().showThoughtInProgress as boolean);
const renderThinkingAsMarkdown = $derived(config().renderThinkingAsMarkdown as boolean);
const showMessageStats = $derived(config().showMessageStats as boolean);
const hasReasoningError = $derived(
isLastAssistantMessage ? !!agenticLastError(message.convId) : false
@@ -354,16 +353,17 @@
{/snippet}
<div class="agentic-content">
{#if highlightTurns && turnGroups.length > 1}
{#if turnGroups.length > 1}
{#each turnGroups as turn, turnIndex (turnIndex)}
{@const turnStats = message?.timings?.agentic?.perTurn?.[turnIndex]}
<div class="agentic-turn my-2 hover:bg-muted/80 dark:hover:bg-muted/30">
<span class="agentic-turn-label">Turn {turnIndex + 1}</span>
<div class="agentic-turn group/turn grid gap-3 mb-4">
{#each turn.sections as section, sIdx (turn.flatIndices[sIdx])}
{@render renderSection(section, turn.flatIndices[sIdx])}
{/each}
{#if turnStats}
<div class="turn-stats">
{#if turnStats && showMessageStats}
<div class="turn-stats transition-opacity duration-150">
<ChatMessageStatistics
promptTokens={turnStats.llm.prompt_n}
promptMs={turnStats.llm.prompt_ms}
@@ -402,39 +402,21 @@
.agentic-content {
display: flex;
flex-direction: column;
gap: 0.5rem;
width: 100%;
max-width: 48rem;
gap: 1rem;
}
.agentic-content > :global(*),
.agentic-turn > :global(*) {
min-width: 0;
}
.agentic-text {
width: 100%;
}
.agentic-turn {
position: relative;
border: 1.5px dashed var(--muted-foreground);
border-radius: 0.75rem;
padding: 1rem;
transition: background 0.1s;
}
.agentic-turn-label {
position: absolute;
top: -1rem;
left: 0.75rem;
padding: 0 0.375rem;
background: var(--background);
font-size: 0.7rem;
font-weight: 500;
color: var(--muted-foreground);
text-transform: uppercase;
letter-spacing: 0.05em;
}
.turn-stats {
margin-top: 0.75rem;
padding-top: 0.5rem;
border-top: 1px solid hsl(var(--muted) / 0.5);
}
</style>
@@ -2,7 +2,7 @@
import { Clock, Gauge, WholeWord, BookOpenText, Sparkles, Wrench, Layers } from '@lucide/svelte';
import { ChatMessageStatisticsBadge } from '$lib/components/app';
import * as Tooltip from '$lib/components/ui/tooltip';
import { ChatMessageStatsView } from '$lib/enums';
import { ChatMessageStatsView, ChatMessageStatisticsMode } from '$lib/enums';
import type { ChatMessageAgenticTimings } from '$lib/types/chat';
import { formatPerformanceTime } from '$lib/utils';
import { MS_PER_SECOND, DEFAULT_PERFORMANCE_TIME } from '$lib/constants';
@@ -19,6 +19,7 @@
agenticTimings?: ChatMessageAgenticTimings;
onActiveViewChange?: (view: ChatMessageStatsView) => void;
hideSummary?: boolean;
mode?: ChatMessageStatisticsMode;
}
let {
@@ -31,19 +32,30 @@
initialView = ChatMessageStatsView.GENERATION,
agenticTimings,
onActiveViewChange,
hideSummary = false
hideSummary = false,
mode = ChatMessageStatisticsMode.SWITCHABLE
}: Props = $props();
let activeView: ChatMessageStatsView = $derived(initialView);
let isSwitchable = $derived(mode === ChatMessageStatisticsMode.SWITCHABLE);
let activeView: ChatMessageStatsView = $derived(
mode === ChatMessageStatisticsMode.READING
? ChatMessageStatsView.READING
: mode === ChatMessageStatisticsMode.GENERATION
? ChatMessageStatsView.GENERATION
: initialView
);
let hasAutoSwitchedToGeneration = $state(false);
$effect(() => {
onActiveViewChange?.(activeView);
if (isSwitchable) {
onActiveViewChange?.(activeView);
}
});
// In live mode: auto-switch to GENERATION tab when prompt processing completes
$effect(() => {
if (isLive) {
if (isLive && isSwitchable) {
// Auto-switch to generation tab only when prompt processing is done (once)
if (
!hasAutoSwitchedToGeneration &&
@@ -91,8 +103,7 @@
formattedPromptTime !== undefined
);
// In live mode, generation tab is disabled until we have generation stats
let isGenerationDisabled = $derived(isLive && !hasGenerationStats);
let isGenerationDisabled = $derived(isLive && isSwitchable && !hasGenerationStats);
let hasAgenticStats = $derived(agenticTimings !== undefined && agenticTimings.toolCallsCount > 0);
@@ -153,44 +164,44 @@
{/snippet}
<div class="inline-flex items-center text-xs text-muted-foreground">
<div class="inline-flex items-center rounded-sm bg-muted-foreground/15 p-0.5">
{#if hasPromptStats || isLive}
{@render viewButton({
view: ChatMessageStatsView.READING,
icon: BookOpenText,
label: 'Reading',
tooltipText: 'Reading (prompt processing)'
})}
{/if}
{@render viewButton({
view: ChatMessageStatsView.GENERATION,
icon: Sparkles,
label: 'Generation',
tooltipText: isGenerationDisabled
? 'Generation (waiting for tokens...)'
: 'Generation (token output)',
disabled: isGenerationDisabled
})}
{#if hasAgenticStats}
{@render viewButton({
view: ChatMessageStatsView.TOOLS,
icon: Wrench,
label: 'Tools',
tooltipText: 'Tool calls'
})}
{#if !hideSummary}
{#if isSwitchable}
<div class="inline-flex items-center rounded-sm bg-muted-foreground/15 p-0.5">
{#if hasPromptStats || isLive}
{@render viewButton({
view: ChatMessageStatsView.SUMMARY,
icon: Layers,
label: 'Summary',
tooltipText: 'Agentic summary'
view: ChatMessageStatsView.READING,
icon: BookOpenText,
label: 'Reading',
tooltipText: 'Processing'
})}
{/if}
{/if}
</div>
{@render viewButton({
view: ChatMessageStatsView.GENERATION,
icon: Sparkles,
label: 'Generation',
tooltipText: isGenerationDisabled ? 'Waiting for tokens...' : 'Generation',
disabled: isGenerationDisabled
})}
{#if hasAgenticStats}
{@render viewButton({
view: ChatMessageStatsView.TOOLS,
icon: Wrench,
label: 'Tools',
tooltipText: 'Tool calls'
})}
{#if !hideSummary}
{@render viewButton({
view: ChatMessageStatsView.SUMMARY,
icon: Layers,
label: 'Summary',
tooltipText: 'Agentic summary'
})}
{/if}
{/if}
</div>
{/if}
<div class="flex items-center gap-1 px-2">
{#if activeView === ChatMessageStatsView.GENERATION && hasGenerationStats}
@@ -256,7 +267,7 @@
value={formattedAgenticTotalTime}
tooltipLabel="Total time (LLM + tools)"
/>
{:else if hasPromptStats}
{:else if hasPromptStats && (mode === ChatMessageStatisticsMode.READING || isSwitchable)}
<ChatMessageStatisticsBadge
class="bg-transparent"
icon={WholeWord}
@@ -186,6 +186,8 @@
message: DatabaseMessage;
toolMessages: DatabaseMessage[];
isLastAssistantMessage: boolean;
isLastUserMessage: boolean;
nextAssistantMessage: DatabaseMessage | null;
siblingInfo: ChatMessageSiblingInfo;
}> = [];
@@ -236,18 +238,36 @@
message: msg,
toolMessages,
isLastAssistantMessage: false,
isLastUserMessage: false,
nextAssistantMessage: null,
siblingInfo
});
}
// Mark the last assistant message
let lastAssistantIdx = -1;
for (let i = result.length - 1; i >= 0; i--) {
if (result[i].message.role === MessageRole.ASSISTANT) {
result[i].isLastAssistantMessage = true;
lastAssistantIdx = i;
break;
}
}
if (lastAssistantIdx > 0 && result[lastAssistantIdx - 1].message.role === MessageRole.USER) {
result[lastAssistantIdx - 1].isLastUserMessage = true;
}
for (let i = 0; i < result.length; i++) {
if (result[i].message.role !== MessageRole.USER) continue;
for (let j = i + 1; j < result.length; j++) {
if (result[j].message.role === MessageRole.ASSISTANT) {
result[i].nextAssistantMessage = result[j].message;
break;
}
}
}
return result;
});
</script>
@@ -257,12 +277,14 @@
{isVisible ? 'opacity-100' : 'opacity-0'}
{previousRouteId === '/(chat)/chat/[id]' ? '' : 'delay-300'}"
>
{#each displayMessages as { message, toolMessages, isLastAssistantMessage, siblingInfo } (message.id)}
{#each displayMessages as { message, toolMessages, isLastAssistantMessage, isLastUserMessage, nextAssistantMessage, siblingInfo } (message.id)}
<ChatMessage
class="mx-auto mt-12 w-full max-w-3xl"
{message}
{toolMessages}
{isLastAssistantMessage}
{isLastUserMessage}
{nextAssistantMessage}
{siblingInfo}
/>
{/each}
@@ -4,12 +4,10 @@
ChatScreenForm,
ChatMessages,
ChatScreenDragOverlay,
ChatScreenProcessingInfo,
ChatScreenStreamResumeStatus,
ServerLoadingSplash,
ChatScreenServerError
} from '$lib/components/app';
import { setProcessingInfoContext } from '$lib/contexts';
import { createAutoScrollController } from '$lib/hooks/use-auto-scroll.svelte';
import { useChatScreenActiveModel } from '$lib/hooks/use-chat-screen-active-model.svelte';
import { useChatScreenDragAndDrop } from '$lib/hooks/use-chat-screen-drag-and-drop.svelte';
@@ -23,8 +21,7 @@
errorDialog,
isLoading,
isChatStreaming,
isEditing,
activeProcessingState
isEditing
} from '$lib/stores/chat.svelte';
import {
conversationsStore,
@@ -42,12 +39,6 @@
let { showCenteredEmpty = false } = $props();
setProcessingInfoContext({
get showProcessingInfo() {
return showProcessingInfo;
}
});
let disableAutoScroll = $derived(Boolean(config().disableAutoScroll) || isMobile.current);
let isMobileUserScrolledUp = $state(false);
let mobileScrollDownHint = $state(false);
@@ -63,11 +54,6 @@
let isServerLoading = $derived(serverLoading());
let hasPropsError = $derived(!!serverError());
let isCurrentConversationLoading = $derived(isLoading() || isChatStreaming());
let showProcessingInfo = $derived(
isCurrentConversationLoading ||
(config().keepStatsVisible && !!page.params.id) ||
activeProcessingState() !== null
);
let chatFormBottomPosition = $derived.by(() => {
if (!isMobile.current) return '1rem';
if (device.isStandalone) return '1.5rem';
@@ -298,10 +284,6 @@
}}
/>
{/if}
{#if showProcessingInfo}
<ChatScreenProcessingInfo />
{/if}
</div>
<ChatScreenForm
@@ -1,127 +0,0 @@
<script lang="ts">
import { untrack } from 'svelte';
import { PROCESSING_INFO_TIMEOUT } from '$lib/constants';
import { useProcessingState } from '$lib/hooks/use-processing-state.svelte';
import { chatStore, isLoading, isChatStreaming } from '$lib/stores/chat.svelte';
import { activeMessages, activeConversation } from '$lib/stores/conversations.svelte';
import { config } from '$lib/stores/settings.svelte';
const processingState = useProcessingState();
let isCurrentConversationLoading = $derived(isLoading());
let isStreaming = $derived(isChatStreaming());
let processingDetails = $derived(processingState.getTechnicalDetails());
let processingVisible = $derived(processingDetails.length > 0);
let { onVisibilityChange }: { onVisibilityChange?: (visible: boolean) => void } = $props();
$effect(() => {
onVisibilityChange?.(processingVisible);
});
$effect(() => {
const conversation = activeConversation();
untrack(() => chatStore.setActiveProcessingConversation(conversation?.id ?? null));
});
$effect(() => {
const keepStatsVisible = config().keepStatsVisible;
const shouldMonitor = keepStatsVisible || isCurrentConversationLoading || isStreaming;
if (shouldMonitor) {
processingState.startMonitoring();
}
if (!isCurrentConversationLoading && !isStreaming && !keepStatsVisible) {
const timeout = setTimeout(() => {
if (!config().keepStatsVisible && !isChatStreaming()) {
processingState.stopMonitoring();
}
}, PROCESSING_INFO_TIMEOUT);
return () => clearTimeout(timeout);
}
});
$effect(() => {
const conversation = activeConversation();
const messages = activeMessages() as DatabaseMessage[];
const keepStatsVisible = config().keepStatsVisible;
if (keepStatsVisible && conversation) {
if (messages.length === 0) {
untrack(() => chatStore.clearProcessingState(conversation.id));
return;
}
if (!isCurrentConversationLoading && !isStreaming) {
untrack(() => chatStore.restoreProcessingStateFromMessages(messages, conversation.id));
}
}
});
</script>
<div
class={[
'chat-processing-info-container pointer-events-none relative w-full hidden md:block',
processingVisible && 'visible'
]}
>
<div class="chat-processing-info-content absolute bottom-4 left-1/2 -translate-x-1/2">
{#each processingDetails as detail (detail)}
<span class="chat-processing-info-detail pointer-events-auto backdrop-blur-sm">{detail}</span>
{/each}
</div>
</div>
<style>
.chat-processing-info-container {
position: sticky;
top: 0;
z-index: 10;
padding: 0 1rem 0.75rem;
opacity: 0;
transform: translateY(50%);
transition:
opacity 300ms ease-out,
transform 300ms ease-out;
}
.chat-processing-info-container.visible {
opacity: 1;
transform: translateY(0);
}
.chat-processing-info-content {
display: flex;
flex-wrap: wrap;
align-items: center;
gap: 1rem;
justify-content: center;
max-width: 48rem;
margin: 0 auto;
}
.chat-processing-info-detail {
color: var(--muted-foreground);
font-size: 0.75rem;
padding: 0.25rem 0.75rem;
border-radius: 0.375rem;
font-family:
ui-monospace, SFMono-Regular, 'SF Mono', Consolas, 'Liberation Mono', Menlo, monospace;
white-space: nowrap;
}
@media (max-width: 768px) {
.chat-processing-info-content {
gap: 0.5rem;
}
.chat-processing-info-detail {
font-size: 0.7rem;
padding: 0.2rem 0.5rem;
}
}
</style>
+9 -12
View File
@@ -241,13 +241,18 @@ export { default as ChatFormActionAddToolsSubmenu } from './ChatForm/ChatFormAct
export { default as ChatFormActionAddMcpServersSubmenu } from './ChatForm/ChatFormActions/ChatFormActionAdd/ChatFormActionAddMcpServersSubmenu.svelte';
/**
* **ChatFormReasoningToggle** - Thinking toggle button with effort dropdown
* Dropdown submenu for selecting reasoning effort level.
*
* A toggle button with lightbulb icon that indicates thinking status.
* Shows the reasoning effort dropdown when clicked.
* Shows a "Reasoning" sub-menu item with a lightbulb icon indicating
* thinking status, and a nested list of effort levels.
* Only visible when the current model supports thinking.
*/
export { default as ChatFormReasoningToggle } from './ChatForm/ChatFormActions/ChatFormReasoningToggle.svelte';
export { default as ChatFormActionAddReasoningSubmenu } from './ChatForm/ChatFormActions/ChatFormActionAdd/ChatFormActionAddReasoningSubmenu.svelte';
/**
* Compact context-usage gauge with per-turn and cumulative breakdown in the tooltip.
*/
export { default as ChatFormContextGauge } from './ChatForm/ChatFormContextGauge/ChatFormContextGauge.svelte';
/**
* Hidden file input element for programmatic file selection.
@@ -669,14 +674,6 @@ export { default as ChatScreenDragOverlay } from './ChatScreen/ChatScreenDragOve
*/
export { default as ChatScreenForm } from './ChatScreen/ChatScreenForm.svelte';
/**
* Processing info display during generation. Shows real-time statistics:
* tokens per second, prompt/completion token counts, and elapsed time.
* Data sourced from slotsService polling during active generation.
* Only visible when `isCurrentConversationLoading` is true.
*/
export { default as ChatScreenProcessingInfo } from './ChatScreen/ChatScreenProcessingInfo.svelte';
/**
* Server error alert displayed when the server is unreachable.
* Shows the error message with a retry button.
@@ -76,7 +76,7 @@
open = value;
onToggle?.();
}}
class={className}
class="{className} my-0!"
>
<Card class="gap-0 border-muted bg-muted/30 py-0">
<Collapsible.Trigger class="flex w-full cursor-pointer items-start justify-between gap-2 p-3">
@@ -72,8 +72,8 @@
</script>
<div
class="code-preview-wrapper rounded-lg border border-border bg-muted {className}"
style="max-height: {maxHeight}; max-width: {maxWidth};"
class="code-preview-wrapper min-w-0 max-w-full overflow-x-auto rounded-lg border border-border bg-muted {className}"
style="max-height: {maxHeight}; {maxWidth ? `max-width: ${maxWidth};` : ''}"
>
<!-- Needs to be formatted as single line for proper rendering -->
<pre class="m-0"><code class="hljs text-sm leading-relaxed">{@html highlightedHtml}</code></pre>
@@ -4,9 +4,10 @@
import * as Dialog from '$lib/components/ui/dialog';
import { fly } from 'svelte/transition';
import { McpServerCardCompact, McpServerForm } from '$lib/components/app/mcp';
import { RECOMMENDED_MCP_SERVERS } from '$lib/constants';
import { RECOMMENDED_MCP_SERVERS, SETTINGS_KEYS } from '$lib/constants';
import { conversationsStore } from '$lib/stores/conversations.svelte';
import { mcpStore } from '$lib/stores/mcp.svelte';
import { settingsStore } from '$lib/stores/settings.svelte';
import { uuid } from '$lib/utils';
import { MCP_SERVERS_ADDED_TO_CHAT_LOCALSTORAGE_KEY, MCP_SERVER_ID_PREFIX } from '$lib/constants';
import type { MCPServerSettingsEntry } from '$lib/types';
@@ -24,6 +25,22 @@
);
let addedServers = $state<MCPServerSettingsEntry[]>([]);
let didAddAny = $state(false);
let selectedRecommendedCount = $derived.by(
() => RECOMMENDED_MCP_SERVERS.filter((server) => selected[server.id]).length
);
let footerLabel = $derived.by(() => {
const recommended = selectedRecommendedCount;
const custom = addedServers.length;
const total = recommended + custom;
if (total === 0) return 'Continue';
if (recommended === 0) return custom === 1 ? 'Add server' : `Add ${custom} servers`;
if (custom === 0) return recommended === 1 ? 'Add server' : `Add ${recommended} servers`;
return `Add ${recommended} servers and ${custom} custom`;
});
let showAddForm = $state(false);
let newServerUrl = $state('');
@@ -44,9 +61,14 @@
showAddForm = false;
newServerUrl = '';
newServerHeaders = '';
addedServers = [];
if (!didAddAny) {
settingsStore.updateConfig(SETTINGS_KEYS.MCP_SERVERS, []);
}
localStorage.setItem(MCP_SERVERS_ADDED_TO_CHAT_LOCALSTORAGE_KEY, 'true');
addedServers = [];
didAddAny = false;
}
open = value;
onOpenChange?.(value);
@@ -59,6 +81,7 @@
}
function enableSelected() {
didAddAny = true;
localStorage.setItem(MCP_SERVERS_ADDED_TO_CHAT_LOCALSTORAGE_KEY, 'true');
for (const server of RECOMMENDED_MCP_SERVERS) {
@@ -83,6 +106,8 @@
function saveNewServer() {
if (newServerUrlError) return;
didAddAny = true;
const newServerId = uuid() ?? `${MCP_SERVER_ID_PREFIX}-${Date.now()}`;
localStorage.setItem(MCP_SERVERS_ADDED_TO_CHAT_LOCALSTORAGE_KEY, 'true');
@@ -174,7 +199,12 @@
<Dialog.Footer>
<Button variant="secondary" size="sm" onclick={() => handleOpenChange(false)}>Not now</Button>
<Button variant="default" size="sm" onclick={enableSelected}>Add selected</Button>
<Button
variant="default"
size="sm"
onclick={enableSelected}
disabled={footerLabel === 'Continue'}>{footerLabel}</Button
>
</Dialog.Footer>
</Dialog.Content>
</Dialog.Root>
@@ -39,13 +39,17 @@
{@const faviconUrl = group.serverId ? mcpStore.getServerFavicon(group.serverId) : null}
<span class="inline-flex min-w-0 items-center gap-1.5 font-medium">
<McpServerIdentity
iconClass="h-4 w-4"
iconRounded="rounded-sm"
showVersion={false}
displayName={group.label}
{faviconUrl}
/>
{#if group.source === 'mcp'}
<McpServerIdentity
iconClass="h-4 w-4"
iconRounded="rounded-sm"
showVersion={false}
displayName={group.label}
{faviconUrl}
/>
{:else}
<TruncatedText text={group.label} class="font-medium" />
{/if}
</span>
<span class="ml-auto shrink-0 text-xs text-muted-foreground">
@@ -0,0 +1,31 @@
<script lang="ts">
import { LinkPreview as HoverCardPrimitive } from 'bits-ui';
import { cn, type WithoutChildrenOrChild } from '$lib/components/ui/utils.js';
import HoverCardPortal from './hover-card-portal.svelte';
import type { ComponentProps } from 'svelte';
let {
ref = $bindable(null),
class: className,
align = 'center',
sideOffset = 4,
portalProps,
...restProps
}: HoverCardPrimitive.ContentProps & {
portalProps?: WithoutChildrenOrChild<ComponentProps<typeof HoverCardPortal>>;
} = $props();
</script>
<HoverCardPortal {...portalProps}>
<HoverCardPrimitive.Content
bind:ref
data-slot="hover-card-content"
{align}
{sideOffset}
class={cn(
'data-open:animate-in data-closed:animate-out data-closed:fade-out-0 data-open:fade-in-0 data-closed:zoom-out-95 data-open:zoom-in-95 data-[side=bottom]:slide-in-from-top-2 data-[side=left]:slide-in-from-right-2 data-[side=right]:slide-in-from-left-2 data-[side=top]:slide-in-from-bottom-2 ring-foreground/10 bg-popover text-popover-foreground w-64 rounded-lg p-2.5 text-sm shadow-md ring-1 duration-100 z-50 origin-(--transform-origin) outline-hidden',
className
)}
{...restProps}
/>
</HoverCardPortal>
@@ -0,0 +1,7 @@
<script lang="ts">
import { LinkPreview as HoverCardPrimitive } from 'bits-ui';
let { ...restProps }: HoverCardPrimitive.PortalProps = $props();
</script>
<HoverCardPrimitive.Portal {...restProps} />
@@ -0,0 +1,7 @@
<script lang="ts">
import { LinkPreview as HoverCardPrimitive } from 'bits-ui';
let { ref = $bindable(null), ...restProps }: HoverCardPrimitive.TriggerProps = $props();
</script>
<HoverCardPrimitive.Trigger bind:ref data-slot="hover-card-trigger" {...restProps} />
@@ -0,0 +1,7 @@
<script lang="ts">
import { LinkPreview as HoverCardPrimitive } from 'bits-ui';
let { open = $bindable(false), ...restProps }: HoverCardPrimitive.RootProps = $props();
</script>
<HoverCardPrimitive.Root bind:open {...restProps} />
@@ -0,0 +1,15 @@
import Root from './hover-card.svelte';
import Content from './hover-card-content.svelte';
import Trigger from './hover-card-trigger.svelte';
import Portal from './hover-card-portal.svelte';
export {
Root,
Content,
Trigger,
Portal,
Root as HoverCard,
Content as HoverCardContent,
Trigger as HoverCardTrigger,
Portal as HoverCardPortal
};
@@ -1,4 +1,3 @@
export const CONTEXT_KEY_MESSAGE_EDIT = 'chat-message-edit';
export const CONTEXT_KEY_CHAT_ACTIONS = 'chat-actions';
export const CONTEXT_KEY_CHAT_SETTINGS_CONFIG = 'chat-settings-config';
export const CONTEXT_KEY_PROCESSING_INFO = 'processing-info';
@@ -17,3 +17,4 @@ export const PANEL_CLASSES = `
`;
export const CHAT_FORM_POPOVER_MAX_HEIGHT = 'max-h-80';
export const DIALOG_SUBMENU_CONTENT = 'w-60';
@@ -6,6 +6,7 @@ import type { ReasoningEffortLevel } from '$lib/types';
* Keys match the ReasoningEffort enum values for type-safe lookups.
*/
export const REASONING_EFFORT_LABELS: Record<string, string> = {
[ReasoningEffort.OFF]: 'Off',
[ReasoningEffort.LOW]: 'Low',
[ReasoningEffort.MEDIUM]: 'Medium',
[ReasoningEffort.HIGH]: 'High',
@@ -13,7 +14,7 @@ export const REASONING_EFFORT_LABELS: Record<string, string> = {
};
export const REASONING_EFFORT_LEVELS: ReasoningEffortLevel[] = [
{ value: 'off', label: 'Off', isOff: true },
{ value: ReasoningEffort.OFF, label: 'Off', isOff: true },
{ value: ReasoningEffort.LOW, label: 'Low' },
{ value: ReasoningEffort.MEDIUM, label: 'Medium' },
{ value: ReasoningEffort.HIGH, label: 'High' },
@@ -22,7 +22,6 @@ export const SETTINGS_KEYS = {
// Display
SHOW_MESSAGE_STATS: 'showMessageStats',
SHOW_THOUGHT_IN_PROGRESS: 'showThoughtInProgress',
KEEP_STATS_VISIBLE: 'keepStatsVisible',
AUTO_MIC_ON_EMPTY: 'autoMicOnEmpty',
RENDER_USER_CONTENT_AS_MARKDOWN: 'renderUserContentAsMarkdown',
DISABLE_AUTO_SCROLL: 'disableAutoScroll',
@@ -61,7 +60,6 @@ export const SETTINGS_KEYS = {
MCP_REQUEST_TIMEOUT_SECONDS: 'mcpRequestTimeoutSeconds',
MCP_DEFAULT_SERVER_OVERRIDES: 'mcpDefaultServerOverrides',
AGENTIC_MAX_TURNS: 'agenticMaxTurns',
ALWAYS_SHOW_AGENTIC_TURNS: 'alwaysShowAgenticTurns',
AGENTIC_MAX_TOOL_PREVIEW_LINES: 'agenticMaxToolPreviewLines',
SHOW_TOOL_CALL_IN_PROGRESS: 'showToolCallInProgress',
// Performance
@@ -258,18 +258,6 @@ const SETTINGS_REGISTRY: Record<string, SettingsSectionEntry> = {
paramType: SyncableParameterType.BOOLEAN
}
},
{
key: SETTINGS_KEYS.KEEP_STATS_VISIBLE,
label: 'Keep stats visible after generation',
help: 'Keep processing statistics visible after generation finishes.',
defaultValue: true,
type: SettingsFieldType.CHECKBOX,
section: SETTINGS_SECTION_SLUGS.DISPLAY,
sync: {
serverKey: SETTINGS_KEYS.KEEP_STATS_VISIBLE,
paramType: SyncableParameterType.BOOLEAN
}
},
{
key: SETTINGS_KEYS.AUTO_MIC_ON_EMPTY,
label: 'Show microphone on empty input',
@@ -379,18 +367,6 @@ const SETTINGS_REGISTRY: Record<string, SettingsSectionEntry> = {
paramType: SyncableParameterType.BOOLEAN
}
},
{
key: SETTINGS_KEYS.ALWAYS_SHOW_AGENTIC_TURNS,
label: 'Always show agentic turns in conversation',
help: 'Always expand and display agentic loop turns in conversation messages.',
defaultValue: false,
type: SettingsFieldType.CHECKBOX,
section: SETTINGS_SECTION_SLUGS.DISPLAY,
sync: {
serverKey: SETTINGS_KEYS.ALWAYS_SHOW_AGENTIC_TURNS,
paramType: SyncableParameterType.BOOLEAN
}
},
{
key: SETTINGS_KEYS.SHOW_BUILD_VERSION,
label: 'Show build version information',
-1
View File
@@ -21,7 +21,6 @@ export const DISABLED_TOOLS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.disabledTool
/** Disabled tools keyed by stable selection identity, no migration from the name based key */
export const DISABLED_TOOL_KEYS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.disabledToolKeys`;
export const FAVORITE_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.favoriteModels`;
export const THINKING_ENABLED_DEFAULT_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.thinkingEnabledDefault`;
export const REASONING_EFFORT_DEFAULT_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.reasoningEffortDefault`;
/** Set when user has interacted with the MCP server recommendations dialog (checked servers, added custom server, or dismissed) */
export const MCP_SERVERS_ADDED_TO_CHAT_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.mcpServersSetupDone`;
-6
View File
@@ -17,9 +17,3 @@ export {
setChatSettingsConfigContext,
type ChatSettingsConfigContext
} from './chat-settings-config.context';
export {
getProcessingInfoContext,
setProcessingInfoContext,
type ProcessingInfoContext
} from './processing-info.context';
@@ -1,16 +0,0 @@
import { getContext, setContext } from 'svelte';
import { CONTEXT_KEY_PROCESSING_INFO } from '$lib/constants';
export interface ProcessingInfoContext {
readonly showProcessingInfo: boolean;
}
const PROCESSING_INFO_KEY = Symbol.for(CONTEXT_KEY_PROCESSING_INFO);
export function setProcessingInfoContext(ctx: ProcessingInfoContext): ProcessingInfoContext {
return setContext(PROCESSING_INFO_KEY, ctx);
}
export function getProcessingInfoContext(): ProcessingInfoContext {
return getContext(PROCESSING_INFO_KEY);
}
+6
View File
@@ -5,6 +5,12 @@ export enum ChatMessageStatsView {
SUMMARY = 'summary'
}
export enum ChatMessageStatisticsMode {
SWITCHABLE = 'switchable',
READING = 'reading',
GENERATION = 'generation'
}
/**
* Connection state of a streamed completion, drives the resume status indicator.
*/
+1
View File
@@ -10,6 +10,7 @@ export { AgenticSectionType, ContinueIntentKind, ToolCallType } from './agentic.
export {
ChatMessageStatsView,
ChatMessageStatisticsMode,
StreamConnectionState,
ContentPartType,
ConversationSelectionMode,
@@ -3,6 +3,7 @@
* These values are sent to the server and mapped to token budgets.
*/
export enum ReasoningEffort {
OFF = 'off',
LOW = 'low',
MEDIUM = 'medium',
HIGH = 'high',
@@ -0,0 +1,295 @@
/**
* Reactive state for the context usage gauge: resolves the active model,
* fetches its cached props, parses live server stats, and exposes per-turn
* read / fresh / cache / output and cumulative token counts.
*/
import {
modelsStore,
modelOptions,
selectedModelId,
singleModelName
} from '$lib/stores/models.svelte';
import { chatStore } from '$lib/stores/chat.svelte';
import { activeMessages } from '$lib/stores/conversations.svelte';
import { isRouterMode } from '$lib/stores/server.svelte';
import { MessageRole } from '$lib/enums';
import { STATS_UNITS } from '$lib/constants';
import type { ChatMessageTimings, DatabaseMessage } from '$lib/types';
import { useProcessingState } from './use-processing-state.svelte';
import {
colorLevelFromPercent,
type ColorLevel
} from '$lib/components/app/chat/ChatForm/ChatFormContextGauge/context-gauge';
interface LiveStats {
freshTokens: number;
promptTokens: number;
cacheTokens: number;
outputTokens: number;
}
export interface UseContextGaugeReturn {
readonly activeModelId: string | null;
readonly isActiveModelLoaded: boolean;
readonly isActiveModelLoading: boolean;
readonly contextTotal: number | null;
readonly contextUsed: number;
readonly currentRead: number;
readonly currentFresh: number;
readonly currentCache: number;
readonly currentOutput: number;
readonly kvTotal: number;
readonly cumulativeRead: number;
readonly cumulativeOutput: number;
readonly cumulativeCacheTotal: number;
readonly averageTokensPerSecond: number | null;
readonly contextPercent: number | null;
readonly colorLevel: ColorLevel;
readonly transientDetails: string[];
readonly hasAnyUsage: boolean;
loadModel(): Promise<void>;
startMonitoring(): void;
}
function lastAssistantTimings(messages: DatabaseMessage[]): ChatMessageTimings | undefined {
for (let i = messages.length - 1; i >= 0; i--) {
const m = messages[i];
if (m.role === MessageRole.ASSISTANT && m.timings) return m.timings;
}
return undefined;
}
function deriveLiveStats(
state: ReturnType<typeof useProcessingState>['processingState']
): LiveStats | null {
if (!state || (state.status !== 'preparing' && state.status !== 'generating')) {
return null;
}
const promptTokens = state.promptTokens ?? 0;
const cacheTokens = state.cacheTokens ?? 0;
return {
freshTokens: promptTokens,
promptTokens: promptTokens + cacheTokens,
cacheTokens,
outputTokens: state.outputTokensUsed ?? 0
};
}
const TRANSIENT_DETAILS_EXCLUDED_PREFIXES = ['Context:', 'Output:'];
function filterTransientDetails(raw: string[]): string[] {
return raw.filter((detail) => {
if (TRANSIENT_DETAILS_EXCLUDED_PREFIXES.some((prefix) => detail.startsWith(prefix))) {
return false;
}
return !detail.includes(STATS_UNITS.TOKENS_PER_SECOND);
});
}
export function useContextGauge(): UseContextGaugeReturn {
const processingState = useProcessingState();
// Resolve the model the gauge reports context for: explicit selection >
// last assistant model > single-model mode (mirrors useChatScreenActiveModel).
const activeModelId = $derived.by(() => {
if (!isRouterMode()) {
return singleModelName();
}
const selectedId = selectedModelId();
if (selectedId) {
const model = modelOptions().find((m) => m.id === selectedId);
if (model) return model.model;
}
return chatStore.getConversationModel(activeMessages() as DatabaseMessage[]);
});
const isActiveModelLoaded = $derived(
activeModelId !== null && modelsStore.isModelLoaded(activeModelId)
);
const isActiveModelLoading = $derived(
activeModelId !== null && modelsStore.isModelOperationInProgress(activeModelId)
);
// Pull /props on demand so n_ctx surfaces before the first chat request.
$effect(() => {
if (activeModelId && isActiveModelLoaded) {
const cached = modelsStore.getModelProps(activeModelId);
if (!cached) {
void modelsStore.fetchModelProps(activeModelId);
}
}
});
const contextTotal = $derived.by(() => {
void modelsStore.propsCacheVersion;
return activeModelId ? modelsStore.getModelContextSize(activeModelId) : null;
});
const liveStats = $derived(deriveLiveStats(processingState.processingState));
const currentRead = $derived.by(() => {
const timings = lastAssistantTimings(activeMessages() as DatabaseMessage[]);
let read = 0;
if (timings) {
read = (timings.prompt_n ?? 0) + (timings.cache_n ?? 0);
}
// live.promptTokens is already the combined reading (prompt + cache),
// so do not also add live.cacheTokens.
if (liveStats && liveStats.promptTokens > 0) {
read = Math.max(read, liveStats.promptTokens);
}
return read;
});
const currentFresh = $derived.by(() => {
const timings = lastAssistantTimings(activeMessages() as DatabaseMessage[]);
const fresh = timings?.prompt_n ?? 0;
return Math.max(fresh, liveStats?.freshTokens ?? 0);
});
const currentCache = $derived.by(() => {
const timings = lastAssistantTimings(activeMessages() as DatabaseMessage[]);
const cached = timings?.cache_n ?? 0;
if (liveStats && liveStats.promptTokens > 0) {
return Math.max(cached, liveStats.cacheTokens);
}
return cached;
});
const currentOutput = $derived.by(() => {
if (liveStats && liveStats.outputTokens > 0) return liveStats.outputTokens;
const timings = lastAssistantTimings(activeMessages() as DatabaseMessage[]);
return timings?.predicted_n ?? 0;
});
const kvTotal = $derived(currentRead + currentOutput);
const contextUsed = $derived(currentRead + currentOutput);
const cumulative = $derived.by(() => {
const messages = activeMessages() as DatabaseMessage[];
// Agentic sessions stamp the same agentic.llm totals onto every
// assistant message; cache_n is never per-turn so cache_total stays 0.
const agenticMessages = messages.filter(
(m) => m.role === MessageRole.ASSISTANT && m.timings?.agentic?.llm?.predicted_n != null
);
if (agenticMessages.length > 0) {
const llm = agenticMessages[agenticMessages.length - 1].timings!.agentic!.llm;
const output = llm.predicted_n ?? 0;
const outputMs = llm.predicted_ms ?? 0;
const averageTokensPerSecond = outputMs > 0 && output > 0 ? (output / outputMs) * 1000 : null;
return {
read: llm.prompt_n ?? 0,
output,
cacheTotal: 0,
averageTokensPerSecond
};
}
let read = 0;
let output = 0;
let outputMs = 0;
let cacheTotal = 0;
for (const m of messages) {
if (m.role !== MessageRole.ASSISTANT || !m.timings) continue;
read += m.timings.prompt_n ?? 0;
cacheTotal += m.timings.cache_n ?? 0;
output += m.timings.predicted_n ?? 0;
outputMs += m.timings.predicted_ms ?? 0;
}
const averageTokensPerSecond = outputMs > 0 && output > 0 ? (output / outputMs) * 1000 : null;
return { read, output, cacheTotal, averageTokensPerSecond };
});
const contextPercent = $derived.by(() => {
if (contextTotal === null || contextTotal <= 0) return null;
return Math.round((contextUsed / contextTotal) * 100);
});
const colorLevel = $derived(colorLevelFromPercent(contextPercent));
// Drop lines the surrounding Context / Output / speed rows already render.
const transientDetails = $derived(filterTransientDetails(processingState.getTechnicalDetails()));
const hasAnyUsage = $derived(
cumulative.read > 0 ||
cumulative.output > 0 ||
currentRead > 0 ||
currentOutput > 0 ||
cumulative.averageTokensPerSecond !== null ||
transientDetails.length > 0
);
async function loadModel() {
if (!activeModelId || isActiveModelLoading) return;
try {
await modelsStore.loadModel(activeModelId);
} catch {
// toast already surfaced by modelsStore.loadModel
}
}
return {
get activeModelId() {
return activeModelId;
},
get isActiveModelLoaded() {
return isActiveModelLoaded;
},
get isActiveModelLoading() {
return isActiveModelLoading;
},
get contextTotal() {
return contextTotal;
},
get contextUsed() {
return contextUsed;
},
get currentRead() {
return currentRead;
},
get currentFresh() {
return currentFresh;
},
get currentCache() {
return currentCache;
},
get currentOutput() {
return currentOutput;
},
get kvTotal() {
return kvTotal;
},
get cumulativeRead() {
return cumulative.read;
},
get cumulativeOutput() {
return cumulative.output;
},
get cumulativeCacheTotal() {
return cumulative.cacheTotal;
},
get averageTokensPerSecond() {
return cumulative.averageTokensPerSecond;
},
get contextPercent() {
return contextPercent;
},
get colorLevel() {
return colorLevel;
},
get transientDetails() {
return transientDetails;
},
get hasAnyUsage() {
return hasAnyUsage;
},
loadModel,
startMonitoring: () => processingState.startMonitoring()
};
}
@@ -54,11 +54,6 @@ export function useMcpRecommendations() {
// effect, and we must not wipe the timeout that was just scheduled.
if (checked) return;
if (mcpStore.optedInRecommendationIds.size > 0) {
checked = true;
return;
}
const hasRecommendations = mcpStore
.getServers()
.some((server) => RECOMMENDED_MCP_SERVER_IDS.has(server.id));
@@ -1,5 +1,4 @@
import { activeProcessingState } from '$lib/stores/chat.svelte';
import { config } from '$lib/stores/settings.svelte';
import { STATS_UNITS } from '$lib/constants';
import type { ApiProcessingState, LiveProcessingStats, LiveGenerationStats } from '$lib/types';
@@ -46,7 +45,6 @@ export function useProcessingState(): UseProcessingStateReturn {
return activeProcessingState();
});
// Track last known state for keepStatsVisible functionality
$effect(() => {
if (processingState && isMonitoring) {
lastKnownState = processingState;
@@ -88,14 +86,8 @@ export function useProcessingState(): UseProcessingStateReturn {
function stopMonitoring(): void {
if (!isMonitoring) return;
isMonitoring = false;
// Only clear last known state if keepStatsVisible is disabled
const currentConfig = config();
if (!currentConfig.keepStatsVisible) {
lastKnownState = null;
lastKnownProcessingStats = null;
}
isMonitoring = false;
}
function getProcessingMessage(): string {
+2 -1
View File
@@ -340,6 +340,7 @@ export class ChatService {
if (stream && conversationId) {
headers['X-Conversation-Id'] = streamIdentity(conversationId, options.model);
}
const response = await fetch(API_CHAT.COMPLETIONS, {
method: 'POST',
headers,
@@ -1015,7 +1016,7 @@ export class ChatService {
*
* @param response - The fetch Response object containing the JSON data
* @param onComplete - Optional callback invoked when response is successfully parsed
* @param onError - Optional callback invoked if an error occurs during parsing
* @param onError - Optional callback invoked if an error occurs while parsing
* @returns {Promise<string>} Promise that resolves to the generated content string
* @throws {Error} if the response cannot be parsed or is malformed
*/
@@ -564,9 +564,9 @@ const configTypesMigration: Migration = {
const config = JSON.parse(configRaw);
let changed = false;
// Pre-schema configs persisted booleans as the strings "true"/"false", which the
// strict server schema now rejects. Coerce those back to real booleans. No config
// string field holds exactly "true"/"false", so the match is unambiguous.
// Pre-schema configs persisted booleans as "true"/"false" strings; the strict server
// schema rejects them. No config string field holds exactly "true"/"false", so the
// match is unambiguous.
for (const key of Object.keys(config)) {
if (config[key] === 'true') {
config[key] = true;
+1 -1
View File
@@ -477,7 +477,7 @@ class AgenticStore {
conversationId: string;
messages: ApiChatMessageData[];
options: AgenticFlowOptions;
tools: ReturnType<typeof mcpStore.getToolDefinitionsForLLM>;
tools: ReturnType<typeof toolsStore.getEnabledToolsForLLM>;
agenticConfig: AgenticConfig;
callbacks: AgenticFlowCallbacks;
signal?: AbortSignal;
+3 -1
View File
@@ -60,6 +60,7 @@ import {
ErrorDialogType,
MessageRole,
MessageType,
ReasoningEffort,
StreamConnectionState
} from '$lib/enums';
@@ -2334,7 +2335,8 @@ class ChatStore {
if (currentConfig.excludeReasoningFromContext) apiOptions.excludeReasoningFromContext = true;
apiOptions.enableThinking = conversationsStore.getThinkingEnabled();
apiOptions.reasoningEffort = conversationsStore.getReasoningEffort();
const effort = conversationsStore.getReasoningEffort();
if (effort !== ReasoningEffort.OFF) apiOptions.reasoningEffort = effort;
if (hasValue(currentConfig.temperature))
apiOptions.temperature = Number(currentConfig.temperature);
+39 -39
View File
@@ -47,7 +47,6 @@ import {
NON_ALPHANUMERIC_REGEX,
MULTIPLE_UNDERSCORE_REGEX,
SETTINGS_KEYS,
THINKING_ENABLED_DEFAULT_LOCALSTORAGE_KEY,
REASONING_EFFORT_DEFAULT_LOCALSTORAGE_KEY
} from '$lib/constants';
@@ -84,11 +83,16 @@ class ConversationsStore {
/** Pending MCP server overrides for new conversations (before first message) */
pendingMcpServerOverrides = $state<McpServerOverride[]>(ConversationsStore.loadMcpDefaults());
/** Global (non-conversation-specific) thinking toggle default */
pendingThinkingEnabled = $state(ConversationsStore.loadThinkingDefaults());
/** Global (non-conversation-specific) thinking toggle default, derived from reasoning effort */
pendingThinkingEnabled = $state(false);
/** Global (non-conversation-specific) reasoning effort default */
pendingReasoningEffort = $state<ReasoningEffort>(ConversationsStore.loadReasoningEffortDefault());
pendingReasoningEffort = $state<ReasoningEffort | ReasoningEffort.OFF>(
ConversationsStore.loadReasoningEffortDefault()
);
/** Last non-off reasoning effort, restored when re-enabling thinking globally */
private lastNonOffEffort: ReasoningEffort | null = null;
private static loadMcpDefaults(): McpServerOverride[] {
const raw = config()[SETTINGS_KEYS.MCP_DEFAULT_SERVER_OVERRIDES];
@@ -112,35 +116,14 @@ class ConversationsStore {
settingsStore.updateConfig(SETTINGS_KEYS.MCP_DEFAULT_SERVER_OVERRIDES, JSON.stringify(plain));
}
/** Load thinking-enabled default from localStorage */
private static loadThinkingDefaults(): boolean {
if (typeof globalThis.localStorage === 'undefined') return true;
try {
const raw = localStorage.getItem(THINKING_ENABLED_DEFAULT_LOCALSTORAGE_KEY);
if (!raw) return true;
return raw === 'true';
} catch {
return true;
}
}
/** Persist thinking-enabled default to localStorage */
private saveThinkingDefaults(): void {
if (typeof globalThis.localStorage === 'undefined') return;
localStorage.setItem(
THINKING_ENABLED_DEFAULT_LOCALSTORAGE_KEY,
this.pendingThinkingEnabled ? 'true' : 'false'
);
}
/** Load reasoning effort default from localStorage */
private static loadReasoningEffortDefault(): ReasoningEffort {
if (typeof globalThis.localStorage === 'undefined') return ReasoningEffort.MEDIUM;
private static loadReasoningEffortDefault(): ReasoningEffort | ReasoningEffort.OFF {
if (typeof globalThis.localStorage === 'undefined') return ReasoningEffort.OFF;
try {
const raw = localStorage.getItem(REASONING_EFFORT_DEFAULT_LOCALSTORAGE_KEY);
return (raw as ReasoningEffort) || ReasoningEffort.MEDIUM;
return (raw as ReasoningEffort | ReasoningEffort.OFF) || ReasoningEffort.OFF;
} catch {
return ReasoningEffort.MEDIUM;
return ReasoningEffort.OFF;
}
}
@@ -303,10 +286,17 @@ class ConversationsStore {
this.pendingMcpServerOverrides = [];
}
// Inherit global thinking default into the new conversation
conversation.thinkingEnabled = this.pendingThinkingEnabled;
// Inherit global thinking/reasoning defaults into the new conversation
const thinkingEnabled = this.getThinkingEnabled();
conversation.thinkingEnabled = thinkingEnabled;
conversation.reasoningEffort =
this.pendingReasoningEffort === ReasoningEffort.OFF ? undefined : this.pendingReasoningEffort;
await DatabaseService.updateConversation(conversation.id, {
thinkingEnabled: this.pendingThinkingEnabled
thinkingEnabled,
reasoningEffort:
this.pendingReasoningEffort === ReasoningEffort.OFF
? undefined
: this.pendingReasoningEffort
});
this.conversations = [conversation, ...this.conversations];
@@ -332,7 +322,6 @@ class ConversationsStore {
}
this.pendingMcpServerOverrides = [];
this.pendingThinkingEnabled = ConversationsStore.loadThinkingDefaults();
this.activeConversation = conversation;
if (conversation.currNode) {
@@ -363,7 +352,7 @@ class ConversationsStore {
this.activeMessages = [];
// reload defaults so new chats inherit persisted state
this.pendingMcpServerOverrides = ConversationsStore.loadMcpDefaults();
this.pendingThinkingEnabled = ConversationsStore.loadThinkingDefaults();
this.pendingReasoningEffort = ConversationsStore.loadReasoningEffortDefault();
}
/**
@@ -794,9 +783,11 @@ class ConversationsStore {
*/
getThinkingEnabled(): boolean {
if (this.activeConversation) {
return this.activeConversation.thinkingEnabled ?? this.pendingThinkingEnabled;
if (this.activeConversation.thinkingEnabled !== undefined) {
return this.activeConversation.thinkingEnabled;
}
}
return this.pendingThinkingEnabled;
return this.getReasoningEffort() !== ReasoningEffort.OFF;
}
/**
@@ -806,8 +797,17 @@ class ConversationsStore {
*/
async setThinkingEnabled(enabled: boolean): Promise<void> {
if (!this.activeConversation) {
this.pendingThinkingEnabled = enabled;
this.saveThinkingDefaults();
if (enabled) {
const effort = this.lastNonOffEffort ?? ReasoningEffort.LOW;
this.pendingReasoningEffort = effort;
this.saveReasoningEffortDefaults();
} else {
if (this.pendingReasoningEffort !== ReasoningEffort.OFF) {
this.lastNonOffEffort = this.pendingReasoningEffort;
}
this.pendingReasoningEffort = ReasoningEffort.OFF;
this.saveReasoningEffortDefaults();
}
return;
}
@@ -831,7 +831,7 @@ class ConversationsStore {
* Gets the effective reasoning effort for the active conversation.
* Returns the conversation override if set, otherwise the global default.
*/
getReasoningEffort(): ReasoningEffort {
getReasoningEffort(): ReasoningEffort | ReasoningEffort.OFF {
if (this.activeConversation) {
return this.activeConversation.reasoningEffort ?? this.pendingReasoningEffort;
}
+8 -98
View File
@@ -12,21 +12,22 @@
* - Lifecycle management (initialize, shutdown)
* - Multi-server coordination
* - Tool name conflict detection and resolution
* - OpenAI-compatible tool definition generation
* - Automatic tool-to-server routing
* - Health checks
*
* MCP connection state and raw `Tool[]` per server are owned here; the
* OpenAI-compatible wire format for those tools is built in `toolsStore`
* (see {@link toolsStore.mcpEntries} / {@link toolsStore.getEnabledToolsForLLM}).
*
* @see MCPService in services/mcp.service.ts for protocol operations
*/
import { browser } from '$app/environment';
import { SvelteSet } from 'svelte/reactivity';
import { SETTINGS_KEYS } from '$lib/constants';
import { MCPService } from '$lib/services/mcp.service';
import { config, settingsStore } from '$lib/stores/settings.svelte';
import { mcpResourceStore } from '$lib/stores/mcp-resources.svelte';
import { serverStore } from '$lib/stores/server.svelte';
import { conversationsStore } from '$lib/stores/conversations.svelte';
import { mode } from 'mode-watcher';
import {
parseMcpServerSettings,
@@ -40,9 +41,7 @@ import {
HealthCheckStatus,
MCPRefType,
ColorMode,
UrlProtocol,
JsonSchemaType,
ToolCallType
UrlProtocol
} from '$lib/enums';
import {
DEFAULT_CACHE_TTL_MS,
@@ -53,12 +52,10 @@ import {
MCP_RECONNECT_BACKOFF_MULTIPLIER,
MCP_RECONNECT_INITIAL_DELAY,
MCP_RECONNECT_MAX_DELAY,
MCP_RECONNECT_ATTEMPT_TIMEOUT_MS,
RECOMMENDED_MCP_SERVER_IDS
MCP_RECONNECT_ATTEMPT_TIMEOUT_MS
} from '$lib/constants';
import type {
MCPToolCall,
OpenAIToolDefinition,
ServerStatus,
ToolExecutionResult,
MCPClientConfig,
@@ -582,30 +579,10 @@ class MCPStore {
}
/**
* Recommended MCP server IDs the user opted in to via per-chat overrides.
* Single source of truth for "which recommendations has the user accepted",
* shared by the recommendations hook and the visible-servers getter.
*/
get optedInRecommendationIds(): ReadonlySet<string> {
const ids = new SvelteSet<string>();
for (const override of conversationsStore.pendingMcpServerOverrides) {
if (RECOMMENDED_MCP_SERVER_IDS.has(override.serverId) && override.enabled) {
ids.add(override.serverId);
}
}
return ids;
}
/**
* MCP servers selectable in chat-add UIs and the settings page:
* enabled in settings and either non-recommended or explicitly opted in.
* MCP servers selectable in chat-add UIs and the settings page.
*/
get visibleMcpServers(): MCPServerSettingsEntry[] {
const optedIn = this.optedInRecommendationIds;
return this.getServersSorted().filter(
(server) =>
server.enabled && (!RECOMMENDED_MCP_SERVER_IDS.has(server.id) || optedIn.has(server.id))
);
return this.getServersSorted().filter((server) => server.enabled);
}
async ensureInitialized(perChatOverrides?: McpServerOverride[]): Promise<boolean> {
@@ -979,73 +956,6 @@ class MCPStore {
}
}
getToolDefinitionsForLLM(): OpenAIToolDefinition[] {
const tools: OpenAIToolDefinition[] = [];
for (const connection of this.connections.values()) {
for (const tool of connection.tools) {
const rawSchema = (tool.inputSchema as Record<string, unknown>) ?? {
type: JsonSchemaType.OBJECT,
properties: {},
required: []
};
tools.push({
type: ToolCallType.FUNCTION as const,
function: {
name: tool.name,
description: tool.description,
parameters: this.normalizeSchemaProperties(rawSchema)
}
});
}
}
return tools;
}
private normalizeSchemaProperties(schema: Record<string, unknown>): Record<string, unknown> {
if (!schema || typeof schema !== 'object') {
return schema;
}
const normalized = { ...schema };
if (normalized.properties && typeof normalized.properties === 'object') {
const props = normalized.properties as Record<string, Record<string, unknown>>;
const normalizedProps: Record<string, Record<string, unknown>> = {};
for (const [key, prop] of Object.entries(props)) {
if (!prop || typeof prop !== 'object') {
normalizedProps[key] = prop;
continue;
}
const normalizedProp = { ...prop };
if (!normalizedProp.type && normalizedProp.default !== undefined) {
const defaultVal = normalizedProp.default;
if (typeof defaultVal === 'string') normalizedProp.type = 'string';
else if (typeof defaultVal === 'number')
normalizedProp.type = Number.isInteger(defaultVal) ? 'integer' : 'number';
else if (typeof defaultVal === 'boolean') normalizedProp.type = 'boolean';
else if (Array.isArray(defaultVal)) normalizedProp.type = 'array';
else if (typeof defaultVal === 'object' && defaultVal !== null)
normalizedProp.type = 'object';
}
if (normalizedProp.properties)
Object.assign(
normalizedProp,
this.normalizeSchemaProperties(normalizedProp as Record<string, unknown>)
);
if (normalizedProp.items && typeof normalizedProp.items === 'object')
normalizedProp.items = this.normalizeSchemaProperties(
normalizedProp.items as Record<string, unknown>
);
normalizedProps[key] = normalizedProp;
}
normalized.properties = normalizedProps;
}
return normalized;
}
getToolNames(): string[] {
return Array.from(this.toolsIndex.keys());
}
+120 -40
View File
@@ -13,33 +13,6 @@ import {
import { SvelteMap, SvelteSet } from 'svelte/reactivity';
/** Stable selection identity for a tool, shared by the disabled set and the permission store */
function toolKey(source: ToolSource, name: string, serverId?: string): string {
switch (source) {
case ToolSource.MCP:
return serverId ? `mcp-${serverId}:${name}` : `mcp:${name}`;
case ToolSource.CUSTOM:
return `custom:${name}`;
case ToolSource.FRONTEND:
return `frontend:${name}`;
default:
return `builtin:${name}`;
}
}
function mcpDefinition(
name: string,
description: string | undefined,
schema?: Record<string, unknown>
): OpenAIToolDefinition {
return {
type: ToolCallType.FUNCTION,
function: {
name,
description,
parameters: schema ?? { type: JsonSchemaType.OBJECT, properties: {}, required: [] }
}
};
}
class ToolsStore {
private _builtinTools = $state<OpenAIToolDefinition[]>([]);
@@ -77,12 +50,96 @@ class ToolsStore {
}
}
private toolKey(source: ToolSource, name: string, serverId?: string): string {
switch (source) {
case ToolSource.MCP:
return serverId ? `mcp-${serverId}:${name}` : `mcp:${name}`;
case ToolSource.CUSTOM:
return `custom:${name}`;
case ToolSource.FRONTEND:
return `frontend:${name}`;
default:
return `builtin:${name}`;
}
}
private inferTypeFromDefault(value: unknown): string | undefined {
if (typeof value === 'string') return 'string';
if (typeof value === 'boolean') return 'boolean';
if (typeof value === 'number') return Number.isInteger(value) ? 'integer' : 'number';
if (Array.isArray(value)) return 'array';
if (value !== null && typeof value === 'object') return 'object';
return undefined;
}
/**
* Recursively normalize a JSON Schema object: infers `type` from `default`
* for properties / items that omit it, and descends into nested `properties`
* and `items`. Returns a new object -- does not mutate the input.
*/
private normalizeJsonSchema(schema: Record<string, unknown>): Record<string, unknown> {
if (!schema || typeof schema !== 'object') return schema;
const normalized: Record<string, unknown> = { ...schema };
if (normalized.properties && typeof normalized.properties === 'object') {
const props = normalized.properties as Record<string, Record<string, unknown>>;
const normalizedProps: Record<string, Record<string, unknown>> = {};
for (const [key, prop] of Object.entries(props)) {
if (!prop || typeof prop !== 'object') {
normalizedProps[key] = prop;
continue;
}
const normalizedProp: Record<string, unknown> = { ...prop };
if (!normalizedProp.type && normalizedProp.default !== undefined) {
const inferred = this.inferTypeFromDefault(normalizedProp.default);
if (inferred) normalizedProp.type = inferred;
}
if (normalizedProp.properties) {
Object.assign(
normalizedProp,
this.normalizeJsonSchema(normalizedProp as Record<string, unknown>)
);
}
if (normalizedProp.items && typeof normalizedProp.items === 'object') {
normalizedProp.items = this.normalizeJsonSchema(
normalizedProp.items as Record<string, unknown>
);
}
normalizedProps[key] = normalizedProp;
}
normalized.properties = normalizedProps;
}
return normalized;
}
private mcpDefinition(
name: string,
description: string | undefined,
schema?: Record<string, unknown>
): OpenAIToolDefinition {
return {
type: ToolCallType.FUNCTION,
function: {
name,
description,
parameters: schema ?? { type: JsonSchemaType.OBJECT, properties: {}, required: [] }
}
};
}
get builtinTools(): OpenAIToolDefinition[] {
return this._builtinTools;
}
get mcpTools(): OpenAIToolDefinition[] {
return mcpStore.getToolDefinitionsForLLM();
return this.mcpEntries().map((e) => e.definition);
}
get frontendTools(): OpenAIToolDefinition[] {
@@ -124,11 +181,22 @@ class ToolsStore {
for (const [serverId, connection] of connections) {
const serverName = mcpStore.getServerDisplayName(serverId);
for (const tool of connection.tools) {
const schema = (tool.inputSchema as Record<string, unknown>) ?? undefined;
const rawSchema = (tool.inputSchema as Record<string, unknown>) ?? {
type: JsonSchemaType.OBJECT,
properties: {},
required: []
};
out.push({
serverId,
serverName,
definition: mcpDefinition(tool.name, tool.description, schema)
definition: {
type: ToolCallType.FUNCTION,
function: {
name: tool.name,
description: tool.description,
parameters: this.normalizeJsonSchema(rawSchema)
}
}
});
}
}
@@ -138,7 +206,7 @@ class ToolsStore {
out.push({
serverId,
serverName,
definition: mcpDefinition(tool.name, tool.description)
definition: this.mcpDefinition(tool.name, tool.description)
});
}
}
@@ -160,14 +228,18 @@ class ToolsStore {
for (const def of this._builtinTools) {
const name = def.function.name;
push({ source: ToolSource.BUILTIN, key: toolKey(ToolSource.BUILTIN, name), definition: def });
push({
source: ToolSource.BUILTIN,
key: this.toolKey(ToolSource.BUILTIN, name),
definition: def
});
}
for (const def of this.frontendTools) {
const name = def.function.name;
push({
source: ToolSource.FRONTEND,
key: toolKey(ToolSource.FRONTEND, name),
key: this.toolKey(ToolSource.FRONTEND, name),
definition: def
});
}
@@ -178,14 +250,18 @@ class ToolsStore {
source: ToolSource.MCP,
serverId,
serverName,
key: toolKey(ToolSource.MCP, name, serverId),
key: this.toolKey(ToolSource.MCP, name, serverId),
definition
});
}
for (const def of this.customTools) {
const name = def.function.name;
push({ source: ToolSource.CUSTOM, key: toolKey(ToolSource.CUSTOM, name), definition: def });
push({
source: ToolSource.CUSTOM,
key: this.toolKey(ToolSource.CUSTOM, name),
definition: def
});
}
return entries;
@@ -233,7 +309,8 @@ class ToolsStore {
/**
* Enabled tool definitions for sending to the LLM.
* MCP tools keep their normalized schemas from mcpStore.
* MCP tool schemas are normalized here so the wire payload is consistent
* across all four sources (built-in, frontend/sandbox, MCP, custom JSON).
* The API identifies tools by name, so a name is sent at most once.
*/
getEnabledToolsForLLM(): OpenAIToolDefinition[] {
@@ -256,7 +333,8 @@ class ToolsStore {
for (const def of this._builtinTools) take(def);
for (const def of this.frontendTools) take(def);
for (const def of mcpStore.getToolDefinitionsForLLM()) take(def);
// mcpEntries() over mcpStore directly so wire shape stays normalized and aligned with the tools UI.
for (const entry of this.mcpEntries()) take(entry.definition);
for (const def of this.customTools) take(def);
return result;
@@ -308,15 +386,17 @@ class ToolsStore {
const connection = mcpStore.getConnections().get(serverId);
if (!connection) return;
for (const tool of connection.tools) {
this._disabledTools.delete(toolKey(ToolSource.MCP, tool.name, serverId));
this._disabledTools.delete(this.toolKey(ToolSource.MCP, tool.name, serverId));
}
this.persistDisabledTools();
}
toggleGroup(group: ToolGroup): void {
const allEnabled = group.tools.every((t) => this.isToolEnabled(t.key));
const target = !allEnabled;
for (const tool of group.tools) {
this.setToolEnabled(tool.key, !allEnabled);
if (target) this._disabledTools.delete(tool.key);
else this._disabledTools.add(tool.key);
}
this.persistDisabledTools();
}
@@ -332,7 +412,7 @@ class ToolsStore {
tools: { name: string; description?: string }[];
}[] {
const result: ReturnType<ToolsStore['getMcpToolsFromHealthChecks']> = [];
for (const server of mcpStore.getServersSorted().filter((s) => s.enabled)) {
for (const server of mcpStore.visibleMcpServers) {
const health = mcpStore.getHealthCheckState(server.id);
if (health.status === HealthCheckStatus.SUCCESS && health.tools.length > 0) {
result.push({
+1 -10
View File
@@ -111,15 +111,7 @@ function findLeafNodeInMap(
}
/**
* Convenience wrapper around {@link findLeafNodeInMap} for callers that only have
* a flat message array.
*
* Finds the leaf node (message with no children) for a given message branch.
* Traverses down the tree following the last child until reaching a leaf.
*
* @param messages - All messages in the conversation
* @param messageId - Starting message ID to find leaf for
* @returns The leaf node ID, or the original messageId if no children
* Convenience wrapper around {@link findLeafNodeInMap} for callers that have a flat message array.
*/
export function findLeafNode(messages: readonly DatabaseMessage[], messageId: string): string {
const nodeMap = new Map(messages.map((msg) => [msg.id, msg] as const));
@@ -225,7 +217,6 @@ export function getMessageSiblings(
/**
* Builds sibling information for every message in a conversation.
* A single node map is shared across all lookups for O(1) access.
*
* @param messages - All messages in the conversation
* @returns Map of message ID to its sibling information