server : add SWA checkpoints (#15293)

* server : add SWA checkpoints ggml-ci * cont : server clean-up * server : handle state restore fails * llama : add extended llama_state_seq_ API * server : do not make checkpoints if --swa-full ggml-ci * llama : remove flags value for NONE * server : configure number of SWA checkpoints with CLI arg ggml-ci * args : fix scope of new argument
2025-10-28 08:31:25 +00:00 · 2025-08-14 14:59:50 +03:00
parent 3973163bff
commit d32e03f449
15 changed files with 206 additions and 54 deletions
--- a/src/llama-memory-hybrid.h
+++ b/src/llama-memory-hybrid.h
@@ -74,8 +74,8 @@ public:

    // state write/load

-    void state_write(llama_io_write_i & io, llama_seq_id seq_id = -1) const override;
-    void state_read (llama_io_read_i  & io, llama_seq_id seq_id = -1)       override;
+    void state_write(llama_io_write_i & io, llama_seq_id seq_id = -1, llama_state_seq_flags flags = 0) const override;
+    void state_read (llama_io_read_i  & io, llama_seq_id seq_id = -1, llama_state_seq_flags flags = 0)       override;

    //
    // llama_memory_hybrid specific API