#include #include #include #include #include "llama.h" #include "mem.h" #define MODEL_PATH "Qwen2.5-7B-Instruct-Q4_K_M.gguf" #define SYSTEM_PROMPT \ "You are a precise lexicographer. You are defining the EXACT target word provided.\n" \ "Do NOT confuse the word with phonetically or visually similar words.\n\n" \ "Output format strictly as follows:\n" \ " () — \n\n" \ "DEFINITION:\n" \ "\n\n" \ "WHEN TO USE:\n" \ "<1 sentence on when to use>\n\n" \ "EXAMPLES:\n" \ "1. \n" \ "2. \n\n" \ "Do not include intro, markdown, or extra commentary." #define CHATML_FMT \ "<|im_start|>system\n%s<|im_end|>\n" \ "<|im_start|>user\nTarget word: %s\nDefine this specific word:<|im_end|>\n" \ "<|im_start|>assistant\n" #define MAX_TOKENS 300 static int build_prompt(char **buf, const char *word) { char *p; int len, written; if (!buf || !word) return -1; len = snprintf(NULL, 0, CHATML_FMT, SYSTEM_PROMPT, word); if (len <= 0) return -1; p = MALLOC((size_t)len + 1); written = snprintf(p, (size_t)len + 1, CHATML_FMT, SYSTEM_PROMPT, word); if (written <= 0) { free(p); return -1; } *buf = p; return written; } static void process_request(struct llama_model *model, const char *word) { char *prompt; int prompt_len; int n_prompt_tokens, i; llama_token *prompt_tokens; llama_token new_token_id; /* Data structure used to pass tokens into llama_decode() */ struct llama_batch batch; struct llama_context *ctx; struct llama_context_params cparams; struct llama_sampler *smpl; struct llama_sampler_chain_params sparams; prompt = NULL; prompt_len = build_prompt(&prompt, word); if (prompt_len <= 0) { fprintf(stderr, "Error: failed to construct prompt for '%s'\n", word); return; } cparams = llama_context_default_params(); cparams.n_ctx = 512; /* context size in tokens */ cparams.n_threads = 4; ctx = llama_init_from_model(model, cparams); if (!ctx) { fprintf(stderr, "Error: failed to create context\n"); free(prompt); return; } const struct llama_vocab *vocab = llama_model_get_vocab(model); n_prompt_tokens = -llama_tokenize(vocab, prompt, prompt_len, NULL, 0, true, true); if (n_prompt_tokens <= 0) { fprintf(stderr, "Error: tokenization sizing failed\n"); llama_free(ctx); free(prompt); return; } prompt_tokens = MALLOC((size_t)n_prompt_tokens * sizeof(llama_token)); if (llama_tokenize(vocab, prompt, prompt_len, prompt_tokens, n_prompt_tokens, true, true) < 0) { fprintf(stderr, "Error: Tokenization failed\n"); free(prompt_tokens); llama_free(ctx); free(prompt); return; } free(prompt); /* prompt string no longer required */ /* Ingest prompt tokens in one batch (parallelizes matrix * multiplications across tokens in the batch) */ batch = llama_batch_get_one(prompt_tokens, n_prompt_tokens); if (llama_decode(ctx, batch) != 0) { fprintf(stderr, "Error: Prompt evaluation failed\n"); free(prompt_tokens); llama_free(ctx); return; } free(prompt_tokens); /* Generation loop */ sparams = llama_sampler_chain_default_params(); smpl = llama_sampler_chain_init(sparams); llama_sampler_chain_add(smpl, llama_sampler_init_penalties( 64, /* last_n: lookback window (64 is standard) */ 1.1f, /* repeat_penalty */ 0.0f, /* frequency_penalty */ 0.0f /* presence_penalty */ )); /* Pick the top token */ llama_sampler_chain_add(smpl, llama_sampler_init_greedy()); for (i = 0; i < MAX_TOKENS; i++) { /* Model outputs next tokens for every token in the prompt. * We need the one after the last token in the prompt */ new_token_id = llama_sampler_sample(smpl, ctx, -1); /* Check for end-of-generation (EOG) tokens: * EOS: end-of-sequence * EOT: end-of-turn * Generates garbage until token limit hit or context window * exhausted, if omitted */ if (llama_vocab_is_eog(vocab, new_token_id)) break; /* LLMS process words as sub-word tokens. * E.g.: unbelievable -> ["un", "believ", "able"] * 128-byte buffer is sufficient */ char buf[128]; /* Convert numeric token id to printable text */ int n = llama_token_to_piece(vocab, new_token_id, buf, sizeof(buf), 0, true); if (n > 0) { fwrite(buf, 1, (size_t)n, stdout); fflush(stdout); } /* Create batch with 1 token: * Prompt has been evaluated. Here, we generate one token at a time. * See autoregressive (AR), diffusion (dLLM), non-autoregressive (NAR), * speculative/MTP for alternative frameworks. */ batch = llama_batch_get_one(&new_token_id, 1); /* llama_decode(): the CPU-heavy forward pass through the transformer model: * - allocates memory and KV cache * - runs matrix multiplications * - generates logits (output prediction vectors) */ if (llama_decode(ctx, batch) != 0) { fprintf(stderr, "llama_decode failed!\n"); break; } } printf("\n"); fflush(stdout); llama_perf_context_print(ctx); llama_sampler_free(smpl); llama_free(ctx); } int main(int argc , char *argv[]) { struct llama_model *model; struct llama_model_params mparams; if (unveil(MODEL_PATH, "r") == -1) err(1, "unveil %s failed", MODEL_PATH); if (unveil(NULL, NULL) == -1) err(1, "unveil lock failed"); if (pledge("stdio rpath", NULL) == -1) err(1, "initial pledge failed"); if (argc < 2) errx(1, "usage: %s [prompt]", argv[0]); llama_backend_init(); mparams = llama_model_default_params(); mparams.n_gpu_layers = 0; /* force all layers onto CPU */ mparams.load_mode = LLAMA_LOAD_MODE_MMAP; model = llama_model_load_from_file(MODEL_PATH, mparams); if (!model) errx(1, "failed to load model from file %s", MODEL_PATH); if (pledge("stdio", NULL) == -1) err(1, "secondary pledge failed"); const char *word = argv[1]; process_request(model, word); llama_model_free(model); llama_backend_free(); return 0; }