mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-09-19 01:05:09 +02:00
Merge branch 'upstream' into concedo_experimental
# Conflicts: # .github/workflows/ui-publish.yml # CODEOWNERS # docs/ops.md # ggml/CMakeLists.txt # ggml/src/CMakeLists.txt # ggml/src/ggml-hexagon/htp/CMakeLists.txt # ggml/src/ggml-hexagon/htp/argsort-ops.c # scripts/sync-ggml.last # tools/mtmd/tests/test-deepseek-ocr.py
This commit is contained in:
@@ -175,6 +175,15 @@ bool server_http_context::init(const common_params & params) {
|
||||
// Middlewares
|
||||
//
|
||||
|
||||
// Frontend paths - all embedded UI assets
|
||||
static const std::unordered_set<std::string> frontend_paths = []() {
|
||||
std::unordered_set<std::string> paths { "/" };
|
||||
for (const llama_ui_asset & a : llama_ui_get_assets()) {
|
||||
paths.insert("/" + a.name);
|
||||
}
|
||||
return paths;
|
||||
}();
|
||||
|
||||
// Public endpoints - API routes plus all embedded UI assets
|
||||
static const std::unordered_set<std::string> get_public_endpoints = []() {
|
||||
std::unordered_set<std::string> endpoints {
|
||||
@@ -182,11 +191,8 @@ bool server_http_context::init(const common_params & params) {
|
||||
"/v1/health",
|
||||
"/models",
|
||||
"/v1/models",
|
||||
"/",
|
||||
};
|
||||
for (const llama_ui_asset & a : llama_ui_get_assets()) {
|
||||
endpoints.insert("/" + a.name);
|
||||
}
|
||||
endpoints.insert(frontend_paths.begin(), frontend_paths.end());
|
||||
return endpoints;
|
||||
}();
|
||||
|
||||
@@ -239,18 +245,9 @@ bool server_http_context::init(const common_params & params) {
|
||||
|
||||
auto middleware_server_state = [this](const httplib::Request & req, httplib::Response & res) {
|
||||
if (!is_ready.load()) {
|
||||
#if defined(LLAMA_UI_HAS_ASSETS)
|
||||
if (const auto tmp = string_split<std::string>(req.path, '.');
|
||||
req.path == "/" || (!tmp.empty() && tmp.back() == "html")) {
|
||||
if (const llama_ui_asset * a = llama_ui_find_asset("loading.html")) {
|
||||
res.status = 503;
|
||||
res.set_content(reinterpret_cast<const char*>(a->data), a->size, "text/html; charset=utf-8");
|
||||
return false;
|
||||
}
|
||||
if (frontend_paths.count(req.path)) {
|
||||
return true; // frontend asset, allow it to load and show "loading"
|
||||
}
|
||||
#else
|
||||
(void)req;
|
||||
#endif
|
||||
// no endpoints are allowed to be accessed when the server is not ready
|
||||
// this is to prevent any data races or inconsistent states
|
||||
res.status = 503;
|
||||
|
||||
@@ -568,10 +568,16 @@ static void handle_with_catch(const char * name, std::function<void()> func) {
|
||||
}
|
||||
}
|
||||
|
||||
// treat a null value as absent so clients can send null to request the server default
|
||||
static bool has_value(const json & data, const char * n) {
|
||||
auto it = data.find(n);
|
||||
return it != data.end() && !it->is_null();
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void field_num<T>::eval(field_eval_context & ctx, const json & data) {
|
||||
for (const auto & n : name) {
|
||||
if (data.contains(n)) {
|
||||
if (has_value(data, n)) {
|
||||
handle_with_catch(n, [&]() {
|
||||
if (custom_handler) {
|
||||
custom_handler(ctx, data);
|
||||
@@ -593,7 +599,7 @@ void field_num<T>::eval(field_eval_context & ctx, const json & data) {
|
||||
void field_str::eval(field_eval_context & ctx, const json & data) {
|
||||
GGML_ASSERT(custom_handler);
|
||||
for (const auto & n : name) {
|
||||
if (data.contains(n)) {
|
||||
if (has_value(data, n)) {
|
||||
handle_with_catch(n, [&]() {
|
||||
custom_handler(ctx, data);
|
||||
});
|
||||
@@ -604,7 +610,7 @@ void field_str::eval(field_eval_context & ctx, const json & data) {
|
||||
|
||||
void field_bool::eval(field_eval_context & ctx, const json & data) {
|
||||
for (const auto & n : name) {
|
||||
if (data.contains(n)) {
|
||||
if (has_value(data, n)) {
|
||||
handle_with_catch(n, [&]() {
|
||||
if (custom_handler) {
|
||||
custom_handler(ctx, data);
|
||||
@@ -620,7 +626,7 @@ void field_bool::eval(field_eval_context & ctx, const json & data) {
|
||||
void field_json::eval(field_eval_context & ctx, const json & data) {
|
||||
GGML_ASSERT(custom_handler);
|
||||
for (const auto & n : name) {
|
||||
if (data.contains(n)) {
|
||||
if (has_value(data, n)) {
|
||||
handle_with_catch(n, [&]() {
|
||||
custom_handler(ctx, data);
|
||||
});
|
||||
|
||||
+649
-352
File diff suppressed because it is too large
Load Diff
@@ -9,10 +9,10 @@ struct server_tool {
|
||||
bool permission_write = false;
|
||||
|
||||
virtual ~server_tool() = default;
|
||||
virtual json get_definition() = 0;
|
||||
virtual json invoke(json params) = 0;
|
||||
virtual json get_definition() const = 0;
|
||||
virtual json invoke(json params) const = 0;
|
||||
|
||||
json to_json();
|
||||
json to_json() const;
|
||||
};
|
||||
|
||||
struct server_tools {
|
||||
|
||||
+125
@@ -0,0 +1,125 @@
|
||||
import os
|
||||
|
||||
import pytest
|
||||
from utils import *
|
||||
|
||||
server: ServerProcess
|
||||
|
||||
# project root, used as the search directory for grep_search/file_glob_search
|
||||
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "..", ".."))
|
||||
|
||||
# marker for the grep_search test to find in this file
|
||||
GREP_MARKER = "llama_cpp_test_tools_builtin_marker_grep_search"
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def create_server():
|
||||
global server
|
||||
server = ServerPreset.router()
|
||||
server.server_tools = "all"
|
||||
|
||||
|
||||
def call_tool(name: str, params: dict) -> dict:
|
||||
res = server.make_request("POST", "/tools", data={"tool": name, "params": params})
|
||||
assert res.status_code == 200, res.body
|
||||
assert "error" not in res.body, res.body
|
||||
return res.body
|
||||
|
||||
|
||||
def call_tool_expect_error(name: str, params: dict) -> str:
|
||||
res = server.make_request("POST", "/tools", data={"tool": name, "params": params})
|
||||
assert res.status_code == 200, res.body
|
||||
assert "error" in res.body, res.body
|
||||
return res.body["error"]
|
||||
|
||||
|
||||
def test_tools_builtin_grep_search():
|
||||
global server
|
||||
server.start()
|
||||
|
||||
res = call_tool("grep_search", {
|
||||
"path": PROJECT_ROOT,
|
||||
"pattern": GREP_MARKER,
|
||||
"include": "test_tools_builtin.py", # bare pattern -> matches basename at any depth
|
||||
})
|
||||
text = res["plain_text_response"]
|
||||
assert "test_tools_builtin.py" in text
|
||||
assert GREP_MARKER in text
|
||||
assert "Total matches: 1" in text
|
||||
|
||||
|
||||
def test_tools_builtin_read_file():
|
||||
global server
|
||||
server.start()
|
||||
|
||||
this_file = os.path.join(PROJECT_ROOT, "tools", "server", "tests", "unit", "test_tools_builtin.py")
|
||||
res = call_tool("read_file", {"path": this_file})
|
||||
text = res["plain_text_response"]
|
||||
assert GREP_MARKER in text
|
||||
assert "def test_tools_builtin_read_file" in text
|
||||
|
||||
|
||||
def test_tools_builtin_write_then_edit_file():
|
||||
global server
|
||||
server.start()
|
||||
|
||||
log_path = os.path.join(PROJECT_ROOT, "test.log")
|
||||
try:
|
||||
write_res = call_tool("write_file", {"path": log_path, "content": "line1\nline2\nline3\n"})
|
||||
assert write_res["result"] == "file written successfully"
|
||||
|
||||
read_before = call_tool("read_file", {"path": log_path})
|
||||
assert read_before["plain_text_response"] == "line1\nline2\nline3\n"
|
||||
|
||||
edit_res = call_tool("edit_file", {
|
||||
"path": log_path,
|
||||
"edits": [
|
||||
{"old_text": "line2", "new_text": "line2-edited"},
|
||||
{"old_text": "line3\n", "new_text": "line3\nline4\n"},
|
||||
],
|
||||
})
|
||||
assert edit_res["result"] == "file edited successfully"
|
||||
assert edit_res["edits_applied"] == 2
|
||||
|
||||
read_after = call_tool("read_file", {"path": log_path})
|
||||
assert read_after["plain_text_response"] == "line1\nline2-edited\nline3\nline4\n"
|
||||
finally:
|
||||
if os.path.exists(log_path):
|
||||
os.remove(log_path)
|
||||
|
||||
|
||||
def test_tools_builtin_edit_file_rejects_non_unique_old_text():
|
||||
global server
|
||||
server.start()
|
||||
|
||||
log_path = os.path.join(PROJECT_ROOT, "test.log")
|
||||
try:
|
||||
call_tool("write_file", {"path": log_path, "content": "dup\ndup\n"})
|
||||
err = call_tool_expect_error("edit_file", {
|
||||
"path": log_path,
|
||||
"edits": [{"old_text": "dup", "new_text": "changed"}],
|
||||
})
|
||||
assert "unique" in err
|
||||
finally:
|
||||
if os.path.exists(log_path):
|
||||
os.remove(log_path)
|
||||
|
||||
|
||||
def test_tools_builtin_edit_file_rejects_overlapping_edits():
|
||||
global server
|
||||
server.start()
|
||||
|
||||
log_path = os.path.join(PROJECT_ROOT, "test.log")
|
||||
try:
|
||||
call_tool("write_file", {"path": log_path, "content": "line1\nline2\n"})
|
||||
err = call_tool_expect_error("edit_file", {
|
||||
"path": log_path,
|
||||
"edits": [
|
||||
{"old_text": "line1\nline2", "new_text": "a"},
|
||||
{"old_text": "line2", "new_text": "b"},
|
||||
],
|
||||
})
|
||||
assert "overlap" in err
|
||||
finally:
|
||||
if os.path.exists(log_path):
|
||||
os.remove(log_path)
|
||||
@@ -113,6 +113,7 @@ class ServerProcess:
|
||||
ui_mcp_proxy: bool = False
|
||||
backend_sampling: bool = False
|
||||
gcp_compat: bool = False
|
||||
server_tools: str | None = None
|
||||
|
||||
# session variables
|
||||
process: subprocess.Popen | None = None
|
||||
@@ -256,6 +257,8 @@ class ServerProcess:
|
||||
server_args.append("--no-cache-idle-slots")
|
||||
if self.ui_mcp_proxy:
|
||||
server_args.append("--ui-mcp-proxy")
|
||||
if self.server_tools:
|
||||
server_args.extend(["--tools", self.server_tools])
|
||||
if self.backend_sampling:
|
||||
server_args.append("--backend_sampling")
|
||||
if self.gcp_compat:
|
||||
|
||||
Reference in New Issue
Block a user