mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-08-26 06:31:13 +02:00
680a9ae63d
* cmake : introduce semantic versioning (wip) This commit introduces semantic versioning to llama.cpp. * squash! cmake : introduce semantic versioning (wip) * cmake : update test-cmake README notes [no ci] * include libmtmd in output so show its semversioned * ci : add make-release workflow * ci : fix build number check in build-cmake-pkg.yml * examples : remove trailing whitespace * ci : abort if upstream ggml version does not exist * ci : extract step contents into scripts * ci : add GGML_NATIVE=OFF to ubuntu job * examples : remove CI build information from test-cmake [no ci] This commit removes the nightly/release information that I added previously to keep this focused only on using building and installing llama.cpp with cmake and being able to quickly verify changes or troubleshoot issues. * ci : merge scripts into single script * remove -dev-build_number support This commit removes the incremental build number (versioning) support that I added. This was incorrect and we should only use the semver for the version. Releases will be tag a nightly build and package maintainers/managers that build from source can use the tag and it is therefor important that the correct version is reported. So a nightly-build will report the semver without the build number. The build number and commit as availble via cmake and test-cmake has been updated to include an example of using them: ```console $ ./build.sh [test-cmake] version: 0.1.0, build: 10360 (08c69e381) ... ``` Refs: https://github.com/ggml-org/llama.cpp/pull/26839#discussion_r3755836969 * docs: add initial release.md documentation * cmake : clean-up and add LLAMA_BUILD_IS_DEV option * ci : remove version input from make-release job * ci : add LLAMA_BUILD_IS_DEV=OFF to build-cmake-pkg.yml Refs: https://github.com/danbev/llama.cpp/actions/runs/31576801921/job/94050639145 * docs : update release notes with LLAMA_BUILD_IS_DEV info [no ci] * ci : add TODO to winget workflow [no ci] --------- Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
149 lines
4.9 KiB
C++
149 lines
4.9 KiB
C++
#include "build-info.h"
|
|
|
|
#include "llama.h"
|
|
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <string>
|
|
#include <vector>
|
|
|
|
// embedded data generated by cmake
|
|
extern const char * LICENSES[];
|
|
|
|
// visible
|
|
int llama_server(int argc, char ** argv);
|
|
int llama_cli(int argc, char ** argv);
|
|
|
|
// hidden
|
|
int llama_completion(int argc, char ** argv);
|
|
int llama_bench(int argc, char ** argv);
|
|
int llama_batched_bench(int argc, char ** argv);
|
|
int llama_fit_params(int argc, char ** argv);
|
|
int llama_quantize(int argc, char ** argv);
|
|
int llama_perplexity(int argc, char ** argv);
|
|
int llama_download(int argc, char ** argv);
|
|
|
|
// Self-update is only supported for binaries built with llama-install.sh
|
|
static int llama_update(int argc, char ** argv) {
|
|
(void) argc;
|
|
(void) argv;
|
|
|
|
#ifdef LLAMA_INSTALL_BUILD
|
|
#if defined(_WIN32)
|
|
return system("powershell -NoProfile -ExecutionPolicy Bypass -Command \"irm https://llama.app/install.ps1 | iex\"");
|
|
#else
|
|
return system("curl -fsSL https://llama.app/install.sh | sh");
|
|
#endif
|
|
#else
|
|
printf("Updates are available only when installed from https://llama.app\n");
|
|
return 1;
|
|
#endif
|
|
}
|
|
|
|
static const char * progname;
|
|
|
|
static int help(int argc, char ** argv);
|
|
static int version(int argc, char ** argv);
|
|
static int licenses(int argc, char ** argv);
|
|
|
|
struct command {
|
|
const char * name;
|
|
const char * desc;
|
|
std::vector<std::string> aliases;
|
|
bool hidden;
|
|
int (*func)(int, char **);
|
|
bool flags = false; // allow --name
|
|
};
|
|
|
|
#ifdef LLAMA_INSTALL_BUILD
|
|
#define UPDATE_HIDDEN false
|
|
#else
|
|
#define UPDATE_HIDDEN true
|
|
#endif
|
|
|
|
static const command cmds[] = {
|
|
{"serve", "HTTP API server", {"server"}, false, llama_server },
|
|
{"cli", "Command-line interactive interface", {"client"}, false, llama_cli },
|
|
{"update", "Update llama to the latest release", {}, UPDATE_HIDDEN, llama_update },
|
|
{"download", "Download a model", {"get"}, false, llama_download },
|
|
{"completion", "Text completion", {"complete"}, true, llama_completion },
|
|
{"bench", "Benchmark prompt processing and text generation", {}, true, llama_bench },
|
|
{"batched-bench", "Benchmark batched decoding performance", {}, true, llama_batched_bench},
|
|
{"fit-params", "Compute parameters to fit a model in device memory", {}, true, llama_fit_params },
|
|
{"quantize", "Quantize a model", {}, true, llama_quantize },
|
|
{"perplexity", "Compute model perplexity and KL divergence", {}, true, llama_perplexity },
|
|
{"version", "Show version", {}, false, version, true },
|
|
{"licenses", "Show third-party licenses", {"credits"}, false, licenses, true },
|
|
{"help", "Show available commands", {}, false, help, true },
|
|
};
|
|
|
|
#undef UPDATE_HIDDEN
|
|
|
|
static int version(int /*argc*/, char ** /*argv*/) {
|
|
llama_print_build_info(llama_version());
|
|
return 0;
|
|
}
|
|
|
|
static int licenses(int /*argc*/, char ** /*argv*/) {
|
|
for (int i = 0; LICENSES[i]; ++i) {
|
|
printf("%s\n", LICENSES[i]);
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
static int help(int argc, char ** argv) {
|
|
const bool show_all = argc >= 2 && std::string(argv[1]) == "all";
|
|
|
|
printf("Usage: %s <command> [options]\n\nAvailable commands:\n", progname);
|
|
|
|
for (const auto & cmd : cmds) {
|
|
if (show_all || !cmd.hidden) {
|
|
printf(" %-15s %s\n", cmd.name, cmd.desc);
|
|
}
|
|
}
|
|
printf("\n");
|
|
|
|
if (!show_all) {
|
|
printf("Run '%s help all' to show additional commands.\n", progname);
|
|
}
|
|
printf("Run '%s <command> --help' for command-specific usage.\n", progname);
|
|
|
|
return 0;
|
|
}
|
|
|
|
static bool matches(std::string arg, const command & cmd) {
|
|
if (cmd.flags && arg.size() > 2 && arg[0] == '-' && arg[1] == '-') {
|
|
arg.erase(0, 2);
|
|
}
|
|
if (arg == cmd.name) {
|
|
return true;
|
|
}
|
|
for (const auto & alias : cmd.aliases) {
|
|
if (arg == alias) {
|
|
return true;
|
|
}
|
|
}
|
|
return false;
|
|
}
|
|
|
|
int main(int argc, char ** argv) {
|
|
progname = argv[0];
|
|
|
|
const std::string arg = argc >= 2 ? argv[1] : "help";
|
|
|
|
for (const auto & cmd : cmds) {
|
|
if (matches(arg, cmd)) {
|
|
// keep cmd.name so the router's child processes re-invoke correctly
|
|
#ifdef _WIN32
|
|
_putenv_s("LLAMA_APP_CMD", cmd.name);
|
|
#else
|
|
setenv("LLAMA_APP_CMD", cmd.name, 1);
|
|
#endif
|
|
return cmd.func(argc - 1, argv + 1);
|
|
}
|
|
}
|
|
|
|
fprintf(stderr, "error: unknown command '%s'\n", arg.c_str());
|
|
return 1;
|
|
}
|