diff options
| -rw-r--r-- | .gitignore | 6 | ||||
| -rw-r--r-- | Makefile | 136 | ||||
| -rw-r--r-- | llm.c | 329 | ||||
| -rw-r--r-- | llm.h | 11 | ||||
| -rw-r--r-- | main.c | 66 | ||||
| -rw-r--r-- | mem.h | 45 |
6 files changed, 593 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..91943f8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +*.o +*.gguf +victor +deps/ + +**/*.swp diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..4f9c63c --- /dev/null +++ b/Makefile @@ -0,0 +1,136 @@ +CC = cc +CXX = c++ +CFLAGS = -march=native -O2 -std=c11 -Wall -Wextra +CXXFLAGS = -march=native -O2 -O2 -Wall -Wextra +INCLUDES = -I. -Ideps/include + +LLAMA_TAG = b10107 +LLAMA_URL = https://github.com/ggml-org/llama.cpp/archive/refs/tags/$(LLAMA_TAG).tar.gz +LLAMA_TAR = deps/llama-$(LLAMA_TAG).tar.gz +LLAMA_SRC_DIR = deps/llama.cpp-src +LLAMA_BUILD = $(LLAMA_SRC_DIR)/build + +STATIC_LIBS = deps/lib/libllama.a \ + deps/lib/libggml.a \ + deps/lib/libggml-cpu.a \ + +# System flags required for static linking llama on OpenBSD +SYS_LIBS = -static -pthread -lm -lc++ -lc++abi + +TARGET = victor +SRC = llm.c main.c +OBJS = $(SRC:.c=.o) + +# Standard tools and target directory +BINDIR = /usr/local/bin +RCDIR = /etc/rc.d +VARDIR = /var/$(TARGET) +INSTALL = install +INSTALL_PROGRAM = $(INSTALL) -m 755 +INSTALL_SCRIPT = $(INSTALL) -m 555 +MKDIR_P = mkdir -p + +# Service user configuration +SERVICE_USER = _victor +SERVICE_GROUP = _victor + +.PHONY: all clean distclean install install-user uninstall + +all: $(TARGET) + +# Explicit link order: libllama -> libggml-cpu -> libggml -> libggml-base +$(TARGET): deps/lib/libllama.a $(OBJS) + $(CXX) $(CXXFLAGS) $(OBJS) \ + -Wl,--start-group \ + deps/lib/libllama.a \ + deps/lib/libggml-cpu.a \ + deps/lib/libggml.a \ + deps/lib/libggml-base.a \ + -Wl,--end-group \ + $(SYS_LIBS) -o $@ + +# BSD suffix rule: compiles any .c file into .o +.c.o: + $(CC) $(CFLAGS) $(INCLUDES) -c $< -o $@ + +# Rebuild objects if the shared header changes +$(OBJS): llm.h + +# Download llama source and build statically +deps/lib/libllama.a deps/lib/libggml.a deps/lib/libggml-cpu.a: + @command -v cmake >/dev/null 2>&1 || { echo >&2 "===> Error: cmake is required but not installed. Aborting."; exit 1; } + @mkdir -p deps/lib deps/include + @echo "===> Downloading llama.cpp $(LLAMA_TAG) source..." + ftp -o $(LLAMA_TAR) $(LLAMA_URL) + + @echo "===> Extracting source archive..." + tar -xzf $(LLAMA_TAR) -C deps + rm -rf $(LLAMA_SRC_DIR) + mv deps/llama.cpp-$(LLAMA_TAG) $(LLAMA_SRC_DIR) + rm -f $(LLAMA_TAR) + + @echo "===> Compiling llama static archives..." + cmake -S $(LLAMA_SRC_DIR) -B $(LLAMA_BUILD) \ + -DCMAKE_BUILD_TYPE=Release \ + -DBUILD_SHARED_LIBS=OFF \ + -DGGML_OPENMP=OFF \ + -DGGML_VULKAN=OFF \ + -DLLAMA_BUILD_APP=OFF \ + -DLLAMA_BUILD_COMMON=OFF \ + -DLLAMA_BUILD_TESTS=OFF \ + -DLLAMA_BUILD_UI=OFF \ + -DLLAMA_BUILD_EXAMPLES=OFF \ + -DLLAMA_BUILD_SERVER=OFF \ + -DLLAMA_BUILD_TOOLS=OFF + + cmake --build $(LLAMA_BUILD) --config Release -j$$(sysctl -n hw.ncpu) + + @echo "===> Installing .a files and headers into deps/..." + find $(LLAMA_BUILD) -type f -name "*.a" -exec cp {} deps/lib/ \; + cp -R $(LLAMA_SRC_DIR)/include/* deps/include/ + cp -R $(LLAMA_SRC_DIR)/ggml/include/* deps/include/ + + @echo "===> Cleaning up build directory..." + rm -rf $(LLAMA_SRC_DIR) + +install-user: + @if ! id -u $(SERVICE_USER) >/dev/null 2>&1; then \ + echo "===> Creating service user..."; \ + groupadd $(SERVICE_GROUP) 2>/dev/null || true; \ + useradd -g $(SERVICE_GROUP) -s /sbin/nologin -m -c "victor daemon user" $(SERVICE_USER); \ + fi + +install: install-user + @if [ ! -f $(TARGET) ]; then \ + echo "Error: $(TARGET) executable not found. Please build it first or provide the binary." >&2; \ + exit 1; \ + fi + @$(MKDIR_P) $(BINDIR) + @echo "===> Installing executable to $(BINDIR)/$(TARGET)..." + @$(INSTALL_PROGRAM) $(TARGET) $(BINDIR)/$(TARGET) + @echo "===> Installing RC script to $(RCDIR)/$(TARGET)..." + @$(MKDIR_P) $(RCDIR) + @$(INSTALL_SCRIPT) $(TARGET).rc $(RCDIR)/$(TARGET) + @echo "===> Creating data directory $(VARDIR)..." + @$(MKDIR_P) $(VARDIR) + @chown $(SERVICE_USER):$(SERVICE_GROUP) $(VARDIR) + @chmod 750 $(VARDIR) + @echo "===> Installation complete." + +uninstall: + @echo "===> Stopping service..." + @-rcctl stop $(TARGET) 2>/dev/null || true + @rm -f $(BINDIR)/$(TARGET) + @rm -f $(RCDIR)/$(TARGET) + @echo "Uninstall complete." + @echo "Note: The data directory and service accounts were preserved." + @echo "To remove all remaining data, run:" + @echo " # rm -rf $(VARDIR)" + @echo " # userdel -r $(SERVICE_USER)" + @echo " # groupdel $(SERVICE_GROUP)" + +clean: + rm -f $(TARGET) $(OBJS) + +distclean: clean + rm -rf deps @@ -0,0 +1,329 @@ +#include <stdio.h> +#include <err.h> +#include <string.h> +#include <syslog.h> + +#include "llm.h" +#include "mem.h" + +#define MAX_TOKENS 300 +#define N_CTX 1024 +#define N_THREADS 4 + +#define SYSTEM_PROMPT \ + "You are a dictionary assistant. Define the target word in the context provided.\n" \ + "DO NOT repeat target word in definition or synonyms. Choose one or two of Formality levels.\n\n" \ + "Output format:\n" \ + "<WORD> (<part of speech>) - <Formality: Conversational|Formal|Literary|Figurative|Archaic>\n\n" \ + "DEFINITION:\n" \ + "<terse, highly accurate definition in context>\n\n" \ + "GENERAL TONE:\n" \ + "<One terse sentence describing general tone>\n\n" \ + "SYNONYMS:\n" \ + "<Three to four standard conversational synonyms>\n\n" \ + "EXAMPLES:\n" \ + "1. <example 1>\n" \ + "2. <example 2>\n\n" + +struct llm_ctx { + struct llama_model *model; + struct llama_context *ctx; + const struct llama_vocab *vocab; + struct llama_sampler *smpl; + + /* Token history of what is currently resident in ctx's KV + * cache for sequence 0, in order. Used to compute a + * longest-common-prefix against each new request so we only + * decode the tail that actually changed. */ + llama_token *cached_tokens; + int n_cached_tokens; + + /* Token count of the system prompt.Used as a sanity-check floor: + * if a later request's LCP against the cache falls below this, the + * system-prompt prefix itself failed to match, which would + * indicate template drift. */ + int n_system_tokens; +}; + +static void llm_log_cb(enum ggml_log_level level, const char *s, void *ud) +{ + (void)ud; + + switch (level) { + case GGML_LOG_LEVEL_WARN: + syslog(LOG_WARNING, "%s", s); + return; + case GGML_LOG_LEVEL_ERROR: + syslog(LOG_ERR, "%s", s); + return; + default: + return; + } +} + +/* Tokenize `text` (already a fully rendered chat-template prompt) into + * a freshly malloc'd buffer. Returns token count via *out_n, or -1 on + * failure (*out is left untouched). Caller owns *out. */ +static int tokenize_prompt(const struct llama_vocab *vocab, + const char *text, int text_len, llama_token **out) +{ + int n; + + n = -llama_tokenize(vocab, text, text_len, NULL, 0, false, true); + if (n <= 0) + return -1; + + *out = MALLOC((size_t)n * sizeof(llama_token)); + + if (llama_tokenize(vocab, text, text_len, *out, n, false, true) < 0) { + free(*out); + return -1; + } + + return n; +} + +struct llm_ctx *llm_init(const char *model_path) +{ + char *sys_prompt; + int sys_prompt_len; + llama_token *sys_tokens; + + struct llm_ctx *llm; + struct llama_model_params mparams; + struct llama_context_params cparams; + struct llama_sampler_chain_params sparams; + struct llama_chat_message sys_msg[1]; + + llama_log_set(llm_log_cb, NULL); + llama_backend_init(); + + mparams = llama_model_default_params(); + mparams.n_gpu_layers = 0; /* force all layers onto CPU */ + + /* LLAMA_LOAD_MODE_MLOCK (force model to RAM, no swap/compression) + * would have been preferrable. But my Raspberry Pi5 doesn't have + * enough memory. */ + mparams.load_mode = LLAMA_LOAD_MODE_NONE; + + llm = MALLOC(sizeof(*llm)); + llm->model = llama_model_load_from_file(model_path, mparams); + if (!llm->model) { + free(llm); + return NULL; + } + + if (!llama_model_chat_template(llm->model, NULL)) + syslog(LOG_WARNING, "model has no embedded chat template"); + + /* Creation of context allocates KV cache and compute buffers. + * Do this once in init, instead of per request. */ + cparams = llama_context_default_params(); + cparams.n_ctx = N_CTX; + cparams.n_threads = N_THREADS; + cparams.n_threads_batch = N_THREADS; + + llm->ctx = llama_init_from_model(llm->model, cparams); + if (!llm->ctx) { + llama_model_free(llm->model); + free(llm); + return NULL; + } + + llm->vocab = llama_model_get_vocab(llm->model); + if (!llm->vocab) { + llama_free(llm->ctx); + llama_model_free(llm->model); + free(llm); + return NULL; + } + + /* Sampler chain, reuse across requests as well. */ + sparams = llama_sampler_chain_default_params(); + llm->smpl = llama_sampler_chain_init(sparams); + + llama_sampler_chain_add(llm->smpl, + llama_sampler_init_penalties( + 64, /* last_n: lookback window */ + 1.1f, /* repeat_penalty */ + 0.0f, /* frequency_penalty */ + 0.0f /* presence_penalty */ + ) + ); + + llama_sampler_chain_add(llm->smpl, llama_sampler_init_greedy()); + + llm->cached_tokens = MALLOC((size_t)N_CTX * sizeof(llama_token)); + llm->n_cached_tokens = 0; + + /* Measure system prompt's token count. Used to check template + * drift in llm_run(), not seed the KV cache. */ + sys_msg[0].role = "system"; + sys_msg[0].content = SYSTEM_PROMPT; + + sys_prompt_len = llama_chat_apply_template(NULL, sys_msg, 1, false, NULL, 0); + if (sys_prompt_len > 0) { + sys_prompt = MALLOC((size_t)sys_prompt_len + 1); + if (llama_chat_apply_template(NULL, sys_msg, 1, false, + sys_prompt, sys_prompt_len + 1) >= 0) { + llm->n_system_tokens = tokenize_prompt(llm->vocab, + sys_prompt, sys_prompt_len, &sys_tokens); + if (llm->n_system_tokens > 0) + free(sys_tokens); + else + llm->n_system_tokens = 0; + } else { + llm->n_system_tokens = 0; + } + free(sys_prompt); + } else { + llm->n_system_tokens = 0; + } + + return llm; +} + +void llm_free(struct llm_ctx *llm) +{ + if (!llm) + return; + llama_sampler_free(llm->smpl); + llama_free(llm->ctx); + llama_model_free(llm->model); + llama_backend_free(); + free(llm->cached_tokens); + free(llm); +} + +void llm_run(struct llm_ctx *llm, const char *user_prompt, FILE *out) +{ + int i, n; + char *prompt; + int prompt_len; + llama_token *new_tokens; + int n_new_tokens; + int n_common; + llama_token new_token_id; + struct llama_batch batch; + + struct llama_chat_message messages[] = { + { "system", SYSTEM_PROMPT }, + { "user", user_prompt } + }; + + /* Retokenize full prompt. Let's expensive than decode; no + * assumptions about the model's internal chat structure. */ + prompt_len = llama_chat_apply_template(NULL, messages, 2, true, NULL, 0); + if (prompt_len <= 0) { + syslog(LOG_ERR, "chat template size calculation failed"); + return; + } + + prompt = MALLOC((size_t)prompt_len + 1); + if (llama_chat_apply_template(NULL, messages, 2, true, prompt, + prompt_len + 1) < 0) { + syslog(LOG_ERR, "failed to apply chat template"); + free(prompt); + return; + } + + n_new_tokens = tokenize_prompt(llm->vocab, prompt, prompt_len, &new_tokens); + free(prompt); + if (n_new_tokens <= 0) { + syslog(LOG_ERR, "tokenization failed"); + return; + } + + if (n_new_tokens + MAX_TOKENS > N_CTX) { + syslog(LOG_ERR, "token count exceeds context window"); + free(new_tokens); + return; + } + + /* llm->cached_tokens contain tokens from the previous request + * (system prompt + any tokens inserted by the model + previous + * user prompt + closing tokens inserted by the model). new_tokens + * should have a similar structure, except the new user prompt. We + * walk the two token sets to locate the first new token: presumably + * the start of the new user prompt. */ + n_common = 0; + while (n_common < llm->n_cached_tokens && n_common < n_new_tokens && + llm->cached_tokens[n_common] == new_tokens[n_common]) + n_common++; + + /* Sanity check */ + if (llm->n_cached_tokens > 0 && n_common < llm->n_system_tokens) { + /* Template may be rendering inconsistently between requests */ + syslog(LOG_WARNING, "cached prefix is shorter than the system prompt"); + } + + /* Force at least one real decode every request. If n_common hit + * n_new_tokens, decode gets skipped entirely, so ctx keeps the + * stale logits from the previous request's EOG -- sampling them + * again just reproduces EOG immediately (empty response on an + * exact repeat). */ + if (n_common == n_new_tokens) + n_common--; + + /* Drop everything in the cache past the common prefix. */ + llama_memory_seq_rm(llama_get_memory(llm->ctx), + 0, /* sequence id: 0 - we only use one sequence */ + n_common, /* start at first differing token */ + -1 /* -1: clear to the end of sequence */ + ); + + if (n_common < n_new_tokens) { + /* Only decode the diff */ + batch = llama_batch_get_one(new_tokens + n_common, + n_new_tokens - n_common); + if (llama_decode(llm->ctx, batch) != 0) { + syslog(LOG_ERR, "prompt evaluation failed"); + free(new_tokens); + llm->n_cached_tokens = n_common; + return; + } + + /* Copy the newly tokens to the buffer */ + memcpy(llm->cached_tokens + n_common, new_tokens + n_common, + (size_t)(n_new_tokens - n_common) * sizeof(llama_token)); + llm->n_cached_tokens = n_new_tokens; + } + + free(new_tokens); + + /* Sampler holds repetition-window state (from the penalties + * stage) across calls since we no longer rebuild it each + * request. Reset that state so one request's repeated words + * don't suppress an unrelated word in the next one. */ + llama_sampler_reset(llm->smpl); + + /* Generation loop */ + for (i = 0; i < MAX_TOKENS; i++) { + new_token_id = llama_sampler_sample(llm->smpl, llm->ctx, -1); + llama_sampler_accept(llm->smpl, new_token_id); + + if (llama_vocab_is_eog(llm->vocab, new_token_id)) + break; + + char buf[128]; + n = llama_token_to_piece(llm->vocab, new_token_id, buf, + sizeof(buf), 0, false); + if (n > 0) { + fwrite(buf, 1, (size_t)n, out); + fflush(out); + } + + batch = llama_batch_get_one(&new_token_id, 1); + if (llama_decode(llm->ctx, batch) != 0) { + syslog(LOG_ERR, "decode() error in generation loop"); + break; + } + + /* New token is now in KV cache. Record it for the next + * request's LCP check. */ + llm->cached_tokens[llm->n_cached_tokens++] = new_token_id; + } + + fprintf(out, "\n\n"); + fflush(out); +} @@ -0,0 +1,11 @@ +#ifndef LLM_H +#define LLM_H + +#include "llama.h" + +struct llm_ctx; +struct llm_ctx *llm_init(const char *model_path); +void llm_run(struct llm_ctx *ctx, const char *prompt, FILE *out); +void llm_free(struct llm_ctx *ctx); + +#endif /* LLM_H */ @@ -0,0 +1,66 @@ +#include <stdlib.h> +#include <err.h> +#include <syslog.h> +#include <unistd.h> + +#include "llm.h" + +#define MODEL_PATH "qwen2.5-3b-instruct-q4_k_m.gguf" + +static void usage(const char *prog) +{ + errx(1, "usage: %s [-m model_path]", prog); +} + +int main(int argc , char *argv[]) +{ + int opt; + const char *prog; + const char *model_path; + struct llm_ctx *llm; + + prog = argv[0]; + model_path = MODEL_PATH; + + openlog(prog, LOG_PID | LOG_NDELAY, LOG_DAEMON); + + while ((opt = getopt(argc, argv, "m:")) != -1) { + switch (opt) { + case 'm': + model_path = optarg; + break; + default: + usage(prog); + } + } + + argc -= optind; + argv += optind; + if (argc != 0) + usage(prog); + + if (unveil(model_path, "r") == -1) { + syslog(LOG_ERR, "unveil %s: %m", model_path); + exit(1); + } + + if (unveil(NULL, NULL) == -1) { + syslog(LOG_ERR, "unveil lock failed: %m"); + exit(1); + } + + if (!(llm = llm_init(model_path))) { + syslog(LOG_ERR, "failed to load model"); + exit(1); + } + + printf("Loaded model\n"); + + /* llm_init() calls mlock, hence pledge() after that. */ + if (pledge("stdio", NULL) == -1) { + syslog(LOG_ERR, "pledge: %m"); + exit(1); + } + + return 0; +} @@ -0,0 +1,45 @@ +#ifndef MEM_H +#define MEM_H + +#include <err.h> +#include <stdlib.h> +#include <syslog.h> + +#define MALLOC(s) xmalloc((s), __FILE__, __LINE__) +#define CALLOC(n, s) xcalloc((n), (s), __FILE__, __LINE__) +#define REALLOC(p, s) xrealloc((p), (s), __FILE__, __LINE__) + +static inline void *xmalloc(size_t s, const char *file, int line) +{ + void *p; + + if (!(p = malloc(s))) { + syslog(LOG_ERR, "malloc: %s (line %d)", file, line); + exit(1); + } + return p; +} + +static inline void *xcalloc(size_t n, size_t s, const char *file, int line) +{ + void *p; + + if (!(p = calloc(n, s))) { + syslog(LOG_ERR, "calloc: %s (line %d)", file, line); + exit(1); + } + return p; +} + +static inline void *xrealloc(void *ptr, size_t s, const char *file, int line) +{ + void *p; + + if (!(p = realloc(ptr, s))) { + syslog(LOG_ERR, "realloc: %s (line %d)", file, line); + exit(1); + } + return p; +} + +#endif /* MEM_H */ |
