From f319cce82a91f85cb3535687ae4e5a68befcc23a Mon Sep 17 00:00:00 2001 From: talksik Date: Sat, 12 Jul 2025 12:22:05 -0700 Subject: [PATCH] integrate llama.cpp for local inference --- .gitmodules | 3 + stream/.gitignore | 4 + stream/Makefile | 11 +++ stream/README.md | 32 +++++++ stream/main.c | 176 ++++++++++++++++++++++++++++++++++++ stream/thirdparty/llama.cpp | 1 + 6 files changed, 227 insertions(+) create mode 100644 .gitmodules create mode 100644 stream/.gitignore create mode 100644 stream/Makefile create mode 100644 stream/README.md create mode 100644 stream/main.c create mode 160000 stream/thirdparty/llama.cpp diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 0000000..46635ba --- /dev/null +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "stream/thirdparty/llama.cpp"] + path = stream/thirdparty/llama.cpp + url = https://github.com/ggml-org/llama.cpp diff --git a/stream/.gitignore b/stream/.gitignore new file mode 100644 index 0000000..a633698 --- /dev/null +++ b/stream/.gitignore @@ -0,0 +1,4 @@ +.cache +stream_app +*.gguf +compile_commands.json diff --git a/stream/Makefile b/stream/Makefile new file mode 100644 index 0000000..8e6ccdb --- /dev/null +++ b/stream/Makefile @@ -0,0 +1,11 @@ +CC = gcc +CFLAGS = -Wall -g -I./thirdparty/llama.cpp/include -I./thirdparty/llama.cpp/ggml/include +TARGET = stream_app +SRC = main.c +LIBS = -L./thirdparty/llama.cpp/build/bin -lllama -lggml -lggml-base + +$(TARGET): $(SRC) + $(CC) $(CFLAGS) -o $(TARGET) $(SRC) $(LIBS) + +clean: + rm -f *.o $(TARGET) diff --git a/stream/README.md b/stream/README.md new file mode 100644 index 0000000..08d03ce --- /dev/null +++ b/stream/README.md @@ -0,0 +1,32 @@ +# flowy.stream + +This is a proof of concept of a showing widgets within a conversation based on intent. + +## Todo +- [x] Integrate llama.cpp with local inference. This will set us up for building many parts of experience. +- [ ] Disect what llama is doing and what the Phi model is doing. +- [ ] Create conversational loop with chat and running context. +- [ ] Generate & render different types of blocks: list, email, doc, etc. +- [ ] Try different models with hugging face +- [ ] Render TUI elements based on commands + +## Dependencies + +### llama.cpp +```sh +#model weights +wget https://huggingface.co/microsoft/Phi-3-mini-4k-instruct-gguf/resolve/main/Phi-3-mini-4k-instruct-q4.gguf +``` + +```sh +cd ./thirdparty/llama.cpp + +rm -rf build +mkdir -p build +cmake -S. -Bbuild +cmake --build build +``` + +The libraries should be in `./thirdparty/llama.cpp/build/bin`. + +NOTE: you may have to disable curl as a flag when configuring llama.cpp via cmake. diff --git a/stream/main.c b/stream/main.c new file mode 100644 index 0000000..033ffba --- /dev/null +++ b/stream/main.c @@ -0,0 +1,176 @@ +#include "thirdparty/llama.cpp/ggml/include/ggml-backend.h" +#include +#include +#include +#include + +typedef enum { + Unspecified, + Question, + Math, + Code, + Email, + Doc, + WebSearch, +} IntentType; + +int main(int argc, char *argv[]) { + printf("Hello world\n"); + + char user_input[500]; + printf("What's up, what can I help with?\n"); + fgets(user_input, sizeof(user_input), stdin); + + char prompt[1000]; + snprintf(prompt, sizeof(prompt), "<|system|>You are a helpful assistant.<|end|>\n<|user|>%s<|end|>\n<|assistant|>", user_input); + + // number of layers to offload to the GPU + int ngl = 99; + // number of tokens to predict + int n_predict = 1000; + + // load dynamic backends + + ggml_backend_load_all(); + + // initialize the model + + struct llama_model_params model_params = llama_model_default_params(); + model_params.n_gpu_layers = ngl; + + struct llama_model *model = llama_model_load_from_file( + "Phi-3-mini-4k-instruct-q4.gguf", model_params); + + if (model == NULL) { + fprintf(stderr, "%s: error: unable to load model\n", __func__); + return 1; + } + + const struct llama_vocab *vocab = llama_model_get_vocab(model); + // tokenize the prompt + + // find the number of tokens in the prompt + const int n_prompt = + -llama_tokenize(vocab, prompt, strlen(prompt), NULL, 0, true, true); + + // allocate space for the tokens and tokenize the prompt + llama_token *prompt_tokens = malloc(n_prompt * sizeof(llama_token)); + if (prompt_tokens == NULL) { + fprintf(stderr, "%s: error: failed to allocate memory for prompt tokens\n", + __func__); + return 1; + } + if (llama_tokenize(vocab, prompt, strlen(prompt), prompt_tokens, n_prompt, + true, true) < 0) { + fprintf(stderr, "%s: error: failed to tokenize the prompt\n", __func__); + return 1; + } + + // initialize the context + + struct llama_context_params ctx_params = llama_context_default_params(); + // n_ctx is the context size + ctx_params.n_ctx = n_prompt + n_predict - 1; + // n_batch is the maximum number of tokens that can be processed in a single + // call to llama_decode + ctx_params.n_batch = n_prompt; + // enable performance counters + ctx_params.no_perf = false; + + struct llama_context *ctx = llama_init_from_model(model, ctx_params); + + if (ctx == NULL) { + fprintf(stderr, "%s: error: failed to create the llama_context\n", + __func__); + return 1; + } + + // initialize the sampler + + struct llama_sampler_chain_params sparams = + llama_sampler_chain_default_params(); + sparams.no_perf = false; + struct llama_sampler *smpl = llama_sampler_chain_init(sparams); + + llama_sampler_chain_add(smpl, llama_sampler_init_greedy()); + + // print the prompt token-by-token + + for (int i = 0; i < n_prompt; i++) { + char buf[128]; + int n = llama_token_to_piece(vocab, prompt_tokens[i], buf, sizeof(buf), 0, + true); + if (n < 0) { + fprintf(stderr, "%s: error: failed to convert token to piece\n", + __func__); + return 1; + } + buf[n] = '\0'; + printf("%s", buf); + } + + // prepare a batch for the prompt + + llama_batch batch = llama_batch_get_one(prompt_tokens, n_prompt); + + // main loop + + const int64_t t_main_start = ggml_time_us(); + int n_decode = 0; + llama_token new_token_id; + + for (int n_pos = 0; n_pos + batch.n_tokens < n_prompt + n_predict;) { + // evaluate the current batch with the transformer model + if (llama_decode(ctx, batch)) { + fprintf(stderr, "%s : failed to eval, return code %d\n", __func__, 1); + return 1; + } + + n_pos += batch.n_tokens; + + // sample the next token + { + new_token_id = llama_sampler_sample(smpl, ctx, -1); + + // is it an end of generation? + if (llama_vocab_is_eog(vocab, new_token_id)) { + break; + } + + char buf[128]; + int n = + llama_token_to_piece(vocab, new_token_id, buf, sizeof(buf), 0, true); + if (n < 0) { + fprintf(stderr, "%s: error: failed to convert token to piece\n", + __func__); + return 1; + } + buf[n] = '\0'; + printf("%s", buf); + fflush(stdout); + + // prepare the next batch with the sampled token + batch = llama_batch_get_one(&new_token_id, 1); + + n_decode += 1; + } + } + + printf("\n"); + + const int64_t t_main_end = ggml_time_us(); + + fprintf(stderr, "%s: decoded %d tokens in %.2f s, speed: %.2f t/s\n", + __func__, n_decode, (t_main_end - t_main_start) / 1000000.0f, + n_decode / ((t_main_end - t_main_start) / 1000000.0f)); + + fprintf(stderr, "\n"); + llama_perf_sampler_print(smpl); + llama_perf_context_print(ctx); + fprintf(stderr, "\n"); + + llama_sampler_free(smpl); + llama_free(ctx); + llama_model_free(model); + free(prompt_tokens); +} diff --git a/stream/thirdparty/llama.cpp b/stream/thirdparty/llama.cpp new file mode 160000 index 0000000..c31e606 --- /dev/null +++ b/stream/thirdparty/llama.cpp @@ -0,0 +1 @@ +Subproject commit c31e60647def83d671bac5ab5b35579bf25d9aa1