]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
llama : document that only one on-device state can be saved per sequence (#23520)
authorTim Neumann <redacted>
Mon, 25 May 2026 07:29:28 +0000 (09:29 +0200)
committerGitHub <redacted>
Mon, 25 May 2026 07:29:28 +0000 (10:29 +0300)
include/llama.h

index 75095b22d08f8813b8f3e37de48e6a340cc0716c..e8374c53b706189849cfc310c271be2cabe783a0 100644 (file)
@@ -874,7 +874,8 @@ extern "C" {
 // work only with partial states, such as SWA KV cache or recurrent cache (e.g. Mamba)
 #define LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY 1
 
-// keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load)
+// Keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load).
+// Getting the state for a seq_id with this flag invalidates all prior states gotten for that seq_id with this flag.
 #define LLAMA_STATE_SEQ_FLAGS_ON_DEVICE 2
 
     typedef uint32_t llama_state_seq_flags;