talk-llama : sync llama.cpp

ggml-ci
2025-08-17 08:01:51 +02:00 · 2025-05-27 17:08:24 +03:00
parent 546928c33f
commit 26eb48cb08
18 changed files with 1968 additions and 1178 deletions
--- a/examples/talk-llama/llama-cparams.h
+++ b/examples/talk-llama/llama-cparams.h
@ -4,6 +4,8 @@

 #include <cstdint>

+#define LLAMA_MAX_PARALLEL_SEQUENCES 64
+
 struct llama_cparams {
    uint32_t n_ctx;           // context size used during inference
    uint32_t n_batch;