wire it up

This commit is contained in:
Xuan Son Nguyen
2026-09-03 00:22:15 +02:00
parent dda508de49
commit a3f62cc133
12 changed files with 413 additions and 51 deletions
+20
View File
@@ -3491,6 +3491,26 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.ui = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_UI"));
add_opt(common_arg(
{"--connect"},
string_format("open a peer-to-peer tunnel for server using llama-connect (default: %s)", params.server_connect ? "enabled" : "disabled"),
[](common_params & params) {
params.server_connect = true;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CONNECT"));
add_opt(common_arg(
{"--connect-code"}, "CODE",
"manually specify a 40-character code for --connect (default: generate a new one for each run)",
[](common_params & params, const std::string & value) {
std::string val = value;
string_replace_all(val, "-", "");
string_replace_all(val, " ", "");
if (val.size() != 40) {
throw std::invalid_argument(string_format("error: invalid connect code '%s', must be 40 characters\n", value.c_str()));
}
params.server_connect_code = val;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CONNECT_CODE"));
add_opt(common_arg(
{"--embedding", "--embeddings"},
string_format("restrict to only support embedding use case; use only with dedicated embedding models (default: %s)", params.embedding ? "enabled" : "disabled"),
+4
View File
@@ -631,6 +631,10 @@ struct common_params {
int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
// llama-connect params
bool server_connect = false;
std::string server_connect_code = "";
std::string hostname = "127.0.0.1";
std::string public_path = ""; // NOLINT
std::string api_prefix = ""; // NOLINT
+1 -1
View File
@@ -178,7 +178,7 @@
| `--reasoning-effort LEVEL` | reasoning effort level given to the chat template: 'default' to keep the template default,<br/>or a level such as 'minimal', 'low', 'medium', 'high', 'xhigh' or 'max' (default: default)<br/>(env: LLAMA_ARG_REASONING_EFFORT) |
| `--reasoning-budget N` | token budget for thinking: -1 for unrestricted, 0 for immediate end, N>0 for token budget (default: -1)<br/>(env: LLAMA_ARG_THINK_BUDGET) |
| `--reasoning-budget-message MESSAGE` | message injected before the end-of-thinking tag when reasoning budget is exhausted (default: none)<br/>(env: LLAMA_ARG_THINK_BUDGET_MESSAGE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: template default)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: enabled)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--chat-template JINJA_TEMPLATE` | set custom jinja chat template (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE) |
| `--chat-template-file JINJA_TEMPLATE_FILE` | set custom jinja chat template file (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE_FILE) |
| `--skip-chat-parsing, --no-skip-chat-parsing` | force a pure content parser, even if a Jinja template is specified; model will output everything in the content section, including any reasoning and/or tool calls (default: disabled)<br/>(env: LLAMA_ARG_SKIP_CHAT_PARSING) |
+1 -1
View File
@@ -256,7 +256,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
| `--reasoning-effort LEVEL` | reasoning effort level given to the chat template: 'default' to keep the template default,<br/>or a level such as 'minimal', 'low', 'medium', 'high', 'xhigh' or 'max' (default: default)<br/>(env: LLAMA_ARG_REASONING_EFFORT) |
| `--reasoning-budget N` | token budget for thinking: -1 for unrestricted, 0 for immediate end, N>0 for token budget (default: -1)<br/>(env: LLAMA_ARG_THINK_BUDGET) |
| `--reasoning-budget-message MESSAGE` | message injected before the end-of-thinking tag when reasoning budget is exhausted (default: none)<br/>(env: LLAMA_ARG_THINK_BUDGET_MESSAGE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: template default)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: enabled)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--chat-template JINJA_TEMPLATE` | set custom jinja chat template (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE) |
| `--chat-template-file JINJA_TEMPLATE_FILE` | set custom jinja chat template file (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE_FILE) |
| `--skip-chat-parsing, --no-skip-chat-parsing` | force a pure content parser, even if a Jinja template is specified; model will output everything in the content section, including any reasoning and/or tool calls (default: disabled)<br/>(env: LLAMA_ARG_SKIP_CHAT_PARSING) |
+2
View File
@@ -39,6 +39,8 @@ set(TARGET llama-server-impl)
add_library(${TARGET}
server.cpp
server-connect.cpp
server-connect.h
server-http.cpp
server-http.h
server-models.cpp
+3 -1
View File
@@ -209,6 +209,8 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--mcp-servers-json JSON` | experimental: inline JSON with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments (default: none)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_MCP_SERVERS_JSON) |
| `-ag, --agent, -no-ag, --no-agent` | whether to enable CORS proxy and all built-in tools - do not enable in untrusted environments (default: disabled)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_AGENT) |
| `--ui, --webui, --no-ui, --no-webui` | whether to enable the Web UI (default: enabled)<br/>(env: LLAMA_ARG_UI) |
| `--connect` | open a peer-to-peer tunnel for server using llama-connect (default: disabled)<br/>(env: LLAMA_ARG_CONNECT) |
| `--connect-code CODE` | manually specify a 40-character code for --connect (default: generate a new one for each run)<br/>(env: LLAMA_ARG_CONNECT_CODE) |
| `--embedding, --embeddings` | restrict to only support embedding use case; use only with dedicated embedding models (default: disabled)<br/>(env: LLAMA_ARG_EMBEDDINGS) |
| `--rerank, --reranking` | enable reranking endpoint on server (default: disabled)<br/>(env: LLAMA_ARG_RERANKING) |
| `--api-key KEY` | API key to use for authentication, multiple keys can be provided as a comma-separated list (default: none)<br/>(env: LLAMA_API_KEY) |
@@ -236,7 +238,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--reasoning-effort LEVEL` | reasoning effort level given to the chat template: 'default' to keep the template default,<br/>or a level such as 'minimal', 'low', 'medium', 'high', 'xhigh' or 'max' (default: default)<br/>(env: LLAMA_ARG_REASONING_EFFORT) |
| `--reasoning-budget N` | token budget for thinking: -1 for unrestricted, 0 for immediate end, N>0 for token budget (default: -1)<br/>(env: LLAMA_ARG_THINK_BUDGET) |
| `--reasoning-budget-message MESSAGE` | message injected before the end-of-thinking tag when reasoning budget is exhausted (default: none)<br/>(env: LLAMA_ARG_THINK_BUDGET_MESSAGE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: template default)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--reasoning-preserve, --no-reasoning-preserve` | preserve reasoning trace in the full history, not just the last assistant message (default: enabled)<br/>compatible with certain templates having 'supports_preserve_reasoning' capability<br/>example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking<br/>(env: LLAMA_ARG_REASONING_PRESERVE) |
| `--chat-template JINJA_TEMPLATE` | set custom jinja chat template (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE) |
| `--chat-template-file JINJA_TEMPLATE_FILE` | set custom jinja chat template file (default: template taken from model's metadata)<br/>if suffix/prefix are specified, template will be disabled<br/>only commonly used templates are accepted (unless --jinja is set before this flag):<br/>list of built-in templates:<br/>bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr<br/>(env: LLAMA_ARG_CHAT_TEMPLATE_FILE) |
| `--skip-chat-parsing, --no-skip-chat-parsing` | force a pure content parser, even if a Jinja template is specified; model will output everything in the content section, including any reasoning and/or tool calls (default: disabled)<br/>(env: LLAMA_ARG_SKIP_CHAT_PARSING) |
+51
View File
@@ -9,6 +9,7 @@
#include "server-common.h"
#include <filesystem>
#include <random>
#include <sstream>
#include <fstream>
@@ -16,6 +17,56 @@
#include <cstring>
#include <type_traits>
#if defined(_WIN32)
#include <windows.h>
#elif defined(__APPLE__) && defined(__MACH__)
#include <mach-o/dyld.h>
#include <limits.h>
#else
#include <unistd.h>
#endif
std::filesystem::path get_server_exec_path() {
#if defined(_WIN32)
wchar_t buf[32768] = { 0 }; // Large buffer to handle long paths
DWORD len = GetModuleFileNameW(nullptr, buf, _countof(buf));
if (len == 0 || len >= _countof(buf)) {
throw std::runtime_error("GetModuleFileNameW failed or path too long");
}
return std::filesystem::path(buf);
#elif defined(__APPLE__) && defined(__MACH__)
char small_path[PATH_MAX];
uint32_t size = sizeof(small_path);
if (_NSGetExecutablePath(small_path, &size) == 0) {
// resolve any symlinks to get absolute path
try {
return std::filesystem::canonical(std::filesystem::path(small_path));
} catch (...) {
return std::filesystem::path(small_path);
}
} else {
// buffer was too small, allocate required size and call again
std::vector<char> buf(size);
if (_NSGetExecutablePath(buf.data(), &size) == 0) {
try {
return std::filesystem::canonical(std::filesystem::path(buf.data()));
} catch (...) {
return std::filesystem::path(buf.data());
}
}
throw std::runtime_error("_NSGetExecutablePath failed after buffer resize");
}
#else
char path[FILENAME_MAX];
ssize_t count = readlink("/proc/self/exe", path, FILENAME_MAX);
if (count <= 0) {
throw std::runtime_error("failed to resolve /proc/self/exe");
}
return std::filesystem::path(std::string(path, count));
#endif
}
json format_error_response(const std::string & message, const enum error_type type) {
std::string type_str;
int code = 500;
+4
View File
@@ -13,6 +13,7 @@
#include <chrono>
#include <condition_variable>
#include <cinttypes>
#include <filesystem>
#include <functional>
#include <mutex>
#include <queue>
@@ -92,6 +93,9 @@ struct server_grammar_trigger {
json format_error_response(const std::string & message, const enum error_type type);
// path of the running llama-server binary, used to spawn siblings. throws on failure
std::filesystem::path get_server_exec_path();
//
// random string / id
//
+267
View File
@@ -0,0 +1,267 @@
#include "server-connect.h"
#include "server-common.h"
#include "subproc.h"
#include <cctype>
#include <chrono>
#include <exception>
#include <filesystem>
#include <random>
#include <string_view>
#include <system_error>
#include <thread>
#include <vector>
// share code = room code + pass code, must match llama-connect and the Web UI
// ref: https://github.com/ggml-org/llama-connect/blob/master/src/protocol.rs
static constexpr size_t CONNECT_ROOM_CODE_LEN = 8;
static constexpr size_t CONNECT_PASS_CODE_LEN = 32;
static const std::string CONNECT_CODE_CHARS = "ABCDEFGHJKMNPQRSTUVWXYZabcdefghjkmnpqrstuvwxyz23456789";
#if defined(_WIN32)
static const std::string CONNECT_EXE_NAME = "llama-connect.exe";
static constexpr char PATH_SEPARATOR = ';';
#else
static const std::string CONNECT_EXE_NAME = "llama-connect";
static constexpr char PATH_SEPARATOR = ':';
#endif
// how long to wait for the child to notice the closed stdin before killing it
static constexpr int CONNECT_STOP_TIMEOUT_MS = 3000;
// the pass code guards the tunnel, so do not use random_string(): its mt19937 is predictable
static std::string gen_share_code() {
std::random_device rd;
std::uniform_int_distribution<size_t> dist(0, CONNECT_CODE_CHARS.size() - 1);
std::string code(CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN, ' ');
for (char & c : code) {
c = CONNECT_CODE_CHARS[dist(rd)];
}
return code;
}
static std::string format_share_code(const std::string & code) {
std::string out;
for (size_t i = 0; i < code.size(); i += 8) {
if (i > 0) {
out += ' ';
}
out += code.substr(i, 8);
}
return out;
}
// whitespace is tolerated, the Web UI shows the code in blocks and users paste it back
// returns an empty string if the key is not a valid share code
static std::string normalize_share_code(const std::string & key) {
std::string code;
for (char c : key) {
if (!std::isspace((unsigned char) c)) {
code += c;
}
}
if (code.size() != CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN) {
return "";
}
if (code.find_first_not_of(CONNECT_CODE_CHARS) != std::string::npos) {
return "";
}
return code;
}
// llama-connect logs as "[LEVEL] text", forward at the same level so our verbosity filter applies
static void forward_child_log(const std::string & line) {
static const std::pair<std::string_view, ggml_log_level> tags[] = {
{ "[ERROR] ", GGML_LOG_LEVEL_ERROR },
{ "[WARN] ", GGML_LOG_LEVEL_WARN },
{ "[INFO] ", GGML_LOG_LEVEL_INFO },
{ "[DEBUG] ", GGML_LOG_LEVEL_DEBUG },
{ "[TRACE] ", GGML_LOG_LEVEL_DEBUG },
};
// untagged lines are the startup banner, show them as info
ggml_log_level level = GGML_LOG_LEVEL_INFO;
const char * text = line.c_str();
for (const auto & [tag, tag_level] : tags) {
if (string_starts_with(line, tag)) {
level = tag_level;
text += tag.size();
break;
}
}
switch (level) {
case GGML_LOG_LEVEL_ERROR: LOG_ERR("connect | %s", text); break;
case GGML_LOG_LEVEL_WARN: LOG_WRN("connect | %s", text); break;
case GGML_LOG_LEVEL_DEBUG: LOG_DBG("connect | %s", text); break;
default: LOG_INF("connect | %s", text); break;
}
}
static bool path_is_file(const std::filesystem::path & p) {
std::error_code ec;
return std::filesystem::is_regular_file(p, ec);
}
std::string server_connect::find_binary() {
// prefer the copy shipped next to llama-server over an unrelated one in PATH
try {
auto sibling = get_server_exec_path().parent_path() / CONNECT_EXE_NAME;
if (path_is_file(sibling)) {
return sibling.string();
}
} catch (const std::exception & e) {
SRV_WRN("could not resolve the llama-server path (%s), looking for llama-connect in PATH only\n", e.what());
}
const std::string path_env = common_get_env("PATH");
size_t start = 0;
while (start <= path_env.size()) {
size_t end = path_env.find(PATH_SEPARATOR, start);
if (end == std::string::npos) {
end = path_env.size();
}
const std::string dir = path_env.substr(start, end - start);
if (!dir.empty()) {
auto candidate = std::filesystem::path(dir) / CONNECT_EXE_NAME;
if (path_is_file(candidate)) {
return candidate.string();
}
}
start = end + 1;
}
return "";
}
std::string server_connect::unavailable_reason(const common_params & params) {
if (!common_subproc::is_supported()) {
return "this build has subprocess support disabled, rebuild with -DLLAMA_SUBPROCESS=ON";
}
if (!params.server_connect_code.empty() && normalize_share_code(params.server_connect_code).empty()) {
return "--connect-code must be " + std::to_string(CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN)
+ " characters from '" + CONNECT_CODE_CHARS + "'";
}
if (find_binary().empty()) {
return "could not find '" + CONNECT_EXE_NAME + "' next to llama-server or in PATH.\n"
" it is a separate binary: download it from https://github.com/ggml-org/llama-connect/releases\n"
" or build llama.cpp with -DLLAMA_CONNECT=ON to have it fetched automatically";
}
return "";
}
bool server_connect::start(const common_params & params) {
const std::string bin = find_binary();
if (bin.empty()) {
SRV_ERR("%s", "llama-connect binary not found\n");
return false;
}
// already validated by unavailable_reason()
const std::string code = params.server_connect_code.empty()
? gen_share_code()
: normalize_share_code(params.server_connect_code);
// always loopback, params.hostname may be 0.0.0.0 or a unix socket which the child cannot dial
const std::vector<std::string> args = {
bin,
"--host", "127.0.0.1",
"--port", std::to_string(params.port),
"--code", code,
// the kernel closes our end of this pipe even if we are killed without cleanup,
// so the child cannot outlive us
"--exit-on-stdin-eof",
};
proc = std::make_unique<common_subproc>();
const int options = subprocess_option_no_window
| subprocess_option_combined_stdout_stderr
| subprocess_option_inherit_environment;
if (!proc->create(args, options)) {
SRV_ERR("failed to spawn '%s'\n", bin.c_str());
proc.reset();
return false;
}
log_thread = std::thread([this]() {
FILE * out = proc->stdout_file();
if (out == nullptr) {
SRV_ERR("%s", "failed to get stdout of the llama-connect process\n");
return;
}
std::vector<char> buf(4096);
while (fgets(buf.data(), (int) buf.size(), out) != nullptr) {
forward_child_log(buf.data());
}
// EOF means the child is gone
if (!stopping.load(std::memory_order_acquire)) {
SRV_ERR("%s", "llama-connect exited on its own, remote access is no longer available\n");
}
});
SRV_INF("%s", "-----------------\n");
SRV_INF("%s", "remote access is enabled via llama-connect\n");
SRV_INF("share code (enter it in the Web UI under Settings -> Remote Access): %s\n",
format_share_code(code).c_str());
SRV_WRN("%s", "anyone with this code can use this server, do not share it publicly\n");
SRV_INF("%s", "-----------------\n");
return true;
}
void server_connect::stop() {
if (!proc) {
return;
}
SRV_INF("%s", "stopping llama-connect...\n");
stopping.store(true, std::memory_order_release);
proc->close_stdin();
for (int elapsed = 0; elapsed < CONNECT_STOP_TIMEOUT_MS && proc->alive(); elapsed += 100) {
std::this_thread::sleep_for(std::chrono::milliseconds(100));
}
// no-op if the child already exited; also unblocks the log thread by closing its stdout
proc->terminate();
if (log_thread.joinable()) {
try {
log_thread.join();
} catch (const std::system_error & e) {
// ~thread() on a still-joinable thread calls std::terminate, detach instead
SRV_ERR("failed to join the llama-connect log thread: %s\n", e.what());
log_thread.detach();
}
}
proc->join(); // reap the zombie
proc.reset();
}
server_connect::server_connect() = default;
server_connect::~server_connect() {
try {
stop();
} catch (const std::exception & e) {
SRV_ERR("failed to stop llama-connect: %s\n", e.what());
} catch (...) {
SRV_ERR("%s", "failed to stop llama-connect\n");
}
}
+36
View File
@@ -0,0 +1,36 @@
#pragma once
#include "common.h"
#include <atomic>
#include <memory>
#include <string>
#include <thread>
struct common_subproc;
// spawns llama-connect, which exposes this server to a remote browser over WebRTC
// core binary is in a separate project: https://github.com/ggml-org/llama-connect
struct server_connect {
server_connect();
~server_connect();
server_connect(const server_connect &) = delete;
server_connect & operator=(const server_connect &) = delete;
// path of the llama-connect binary, empty if not found
static std::string find_binary();
// why --connect cannot work here, empty if it can
static std::string unavailable_reason(const common_params & params);
bool start(const common_params & params);
// idempotent, also called by the destructor
void stop();
private:
std::unique_ptr<common_subproc> proc;
std::thread log_thread;
std::atomic<bool> stopping{false}; // tells the log thread the exit is expected
};
+2 -48
View File
@@ -24,7 +24,6 @@
#include <atomic>
#include <chrono>
#include <queue>
#include <filesystem>
#include <random>
#include <sstream>
#include <cstring>
@@ -33,12 +32,6 @@
extern char **environ;
#endif
#if defined(__APPLE__) && defined(__MACH__)
// macOS: use _NSGetExecutablePath to get the executable path
#include <mach-o/dyld.h>
#include <limits.h>
#endif
#define DEFAULT_STOP_TIMEOUT 10 // seconds
#define CMD_ROUTER_TO_CHILD_EXIT "cmd_router_to_child:exit"
@@ -256,47 +249,6 @@ struct server_lru_sched {
// delete). distinct from params.timeout_read/write which only applies to the generation proxy
static constexpr int STREAM_LOOKUP_TIMEOUT_MS = 250;
static std::filesystem::path get_server_exec_path() {
#if defined(_WIN32)
wchar_t buf[32768] = { 0 }; // Large buffer to handle long paths
DWORD len = GetModuleFileNameW(nullptr, buf, _countof(buf));
if (len == 0 || len >= _countof(buf)) {
throw std::runtime_error("GetModuleFileNameW failed or path too long");
}
return std::filesystem::path(buf);
#elif defined(__APPLE__) && defined(__MACH__)
char small_path[PATH_MAX];
uint32_t size = sizeof(small_path);
if (_NSGetExecutablePath(small_path, &size) == 0) {
// resolve any symlinks to get absolute path
try {
return std::filesystem::canonical(std::filesystem::path(small_path));
} catch (...) {
return std::filesystem::path(small_path);
}
} else {
// buffer was too small, allocate required size and call again
std::vector<char> buf(size);
if (_NSGetExecutablePath(buf.data(), &size) == 0) {
try {
return std::filesystem::canonical(std::filesystem::path(buf.data()));
} catch (...) {
return std::filesystem::path(buf.data());
}
}
throw std::runtime_error("_NSGetExecutablePath failed after buffer resize");
}
#else
char path[FILENAME_MAX];
ssize_t count = readlink("/proc/self/exe", path, FILENAME_MAX);
if (count <= 0) {
throw std::runtime_error("failed to resolve /proc/self/exe");
}
return std::filesystem::path(std::string(path, count));
#endif
}
static void unset_reserved_args(common_preset & preset, bool unset_model_args) {
preset.unset_option("LLAMA_ARG_SSL_KEY_FILE");
preset.unset_option("LLAMA_ARG_SSL_CERT_FILE");
@@ -305,6 +257,8 @@ static void unset_reserved_args(common_preset & preset, bool unset_model_args) {
preset.unset_option("LLAMA_ARG_MODELS_MAX");
preset.unset_option("LLAMA_ARG_MODELS_PRESET");
preset.unset_option("LLAMA_ARG_MODELS_AUTOLOAD");
preset.unset_option("LLAMA_ARG_CONNECT");
preset.unset_option("LLAMA_ARG_CONNECT_CODE");
if (unset_model_args) {
preset.unset_option("LLAMA_ARG_MODEL");
preset.unset_option("LLAMA_ARG_MMPROJ");
+22
View File
@@ -1,3 +1,4 @@
#include "server-connect.h"
#include "server-context.h"
#include "server-http.h"
#include "server-models.h"
@@ -175,6 +176,15 @@ int llama_server(common_params & params, int argc, char ** argv) {
params.model_alias.insert(model_name);
}
// check early, so a missing llama-connect fails-fast
if (params.server_connect) {
const std::string reason = server_connect::unavailable_reason(params);
if (!reason.empty()) {
SRV_ERR("--connect is not available: %s\n", reason.c_str());
return 1;
}
}
// note: this is guaranteed to out-live ctx_http and tools
server_mcp mcp_mgr;
@@ -514,6 +524,18 @@ int llama_server(common_params & params, int argc, char ** argv) {
SRV_INF("listening on %s\n", ctx_http.listening_address.c_str());
// spawn only once listening, so the child health check passes. the destructor stops it
server_connect connect_proc;
if (params.server_connect && !connect_proc.start(params)) {
SRV_ERR("%s", "exiting due to llama-connect error\n");
ctx_http.stop();
if (ctx_http.thread.joinable()) {
ctx_http.thread.join();
}
clean_up();
return 1;
}
// TODO: remove this in the future
// check the string to also handle the .sock case
if (string_ends_with(ctx_http.listening_address, ":8080")) {