@setoelkahfi / sigit / commits / 51d281a

Add durable sessions and context compaction

Conversation history lived in RAM, so a restart lost everything and the hard cap of 10 tool rounds was the only defense against blowing the context window. Both problems hurt most on-device, where context is smallest. The InferenceBackend trait gains history_snapshot, restore_history, and compact_history. A new session_store module writes one JSONL file per session under $SIGIT_CONFIG_DIR/sessions/, saved after every completed turn on both surfaces. ACP session/load restores from the store, the TUI gets /resume, and /clear deletes the file. Compaction rebuilds history as system prompt, a summary produced by the backend itself, and the last six messages. It runs automatically before a tool round once the estimate crosses 24k tokens, and on demand via /compact. A failed summarization rolls history back untouched, and the rebuilt tail drops tool results whose assistant message was summarized away, since strict endpoints reject orphans. With compaction in place MAX_TOOL_ROUNDS rises from 10 to 24.

paydii committed Jul 4, 2026 at 22:31 UTC 51d281add943a0c545c8a74af87c0cf0fbd9db17
4 files changed +656 -6
src/backend.rs
+333 -1
@@ -21,7 +21,7 @@
21 use std::sync::Arc;
22
23 use async_trait::async_trait;
24 -use onde::inference::{ChatEngine, ToolDefinition};
24 +use onde::inference::{ChatEngine, ChatMessage, ChatRole, ToolDefinition};
25 use serde::Deserialize;
26 use tokio::sync::Mutex;
27
@@ -62,6 +62,30 @@ pub struct TurnResult {
62 /// Backend errors are plain strings. Callers map them to ACP errors.
63 pub type BackendError = String;
64
65 +/// Rough context budget for a conversation, in estimated tokens (see
66 +/// [`estimate_tokens`]). When a snapshot exceeds this, the agent loops compact
67 +/// history before the next tool round.
68 +pub const DEFAULT_CONTEXT_TOKEN_BUDGET: usize = 24_000;
69 +
70 +/// How many trailing messages survive a compaction verbatim (the rest are
71 +/// folded into the summary).
72 +pub const COMPACT_KEEP_LAST: usize = 6;
73 +
74 +/// The summarization request sent to the model when compacting history.
75 +const SUMMARIZE_PROMPT: &str = "Summarize this coding session so far: decisions made, \
76 + files touched, current state, open items. Be concise and factual.";
77 +
78 +/// Crude token estimate for a history snapshot: serialized characters / 4.
79 +/// Deliberately model-agnostic — it only needs to be in the right ballpark to
80 +/// decide when compaction is worth an extra inference round.
81 +pub fn estimate_tokens(history: &[serde_json::Value]) -> usize {
82 + let chars: usize = history
83 + .iter()
84 + .map(|message| message.to_string().chars().count())
85 + .sum();
86 + chars / 4
87 +}
88 +
89 /// A sink for streaming assistant text deltas to the UI as they are produced.
90 ///
91 /// When a caller passes `Some(sink)`, a streaming-capable backend forwards each
@@ -110,6 +134,23 @@ pub trait InferenceBackend: Send + Sync {
134 /// than on-device. Drives UI labelling so the displayed model can't claim a
135 /// local model while requests actually go to the cloud.
136 fn is_remote(&self) -> bool;
137 +
138 + /// A serializable snapshot of the conversation history, one JSON object per
139 + /// message (`{"role": ..., "content": ...}` at minimum). The snapshot is
140 + /// what the session store persists; it includes any seeded system message
141 + /// so [`InferenceBackend::restore_history`] can replace state wholesale.
142 + async fn history_snapshot(&self) -> Vec<serde_json::Value>;
143 +
144 + /// Replace the conversation history with a previously saved snapshot.
145 + /// Backends that cannot represent every entry (e.g. on-device history has
146 + /// no tool-call structure) flatten what they can and drop the rest.
147 + async fn restore_history(&self, history: Vec<serde_json::Value>);
148 +
149 + /// Shrink the conversation history: summarize everything so far with one
150 + /// extra (non-streaming) inference round, then rebuild history as
151 + /// `[system message, summary, last keep_last non-system messages]`. On
152 + /// error the original history is left in place.
153 + async fn compact_history(&self, keep_last: usize) -> Result<(), BackendError>;
154 }
155
156 // ── Local backend (onde ChatEngine) ──────────────────────────────────────────────
@@ -213,6 +254,82 @@ impl InferenceBackend for LocalBackend {
254 fn is_remote(&self) -> bool {
255 false
256 }
257 +
258 + async fn history_snapshot(&self) -> Vec<serde_json::Value> {
259 + // onde's `history()` already flattens tool entries: assistant tool
260 + // calls become plain assistant text and tool results are omitted, so
261 + // the snapshot is lossy for tool-heavy turns (acceptable in this MVP).
262 + self.engine
263 + .history()
264 + .await
265 + .iter()
266 + .map(|message| {
267 + serde_json::json!({
268 + "role": message.role.to_string(),
269 + "content": message.content,
270 + })
271 + })
272 + .collect()
273 + }
274 +
275 + async fn restore_history(&self, history: Vec<serde_json::Value>) {
276 + self.engine.clear_history().await;
277 + for entry in history {
278 + let role = entry["role"].as_str().unwrap_or("");
279 + let content = entry["content"].as_str().unwrap_or("").to_string();
280 + // Tool-call-only assistant entries and empty tool results carry no
281 + // text a plain chat history can replay; drop them.
282 + if content.is_empty() && role != "user" && role != "system" {
283 + continue;
284 + }
285 + let message = match role {
286 + "system" => ChatMessage::system(content),
287 + "user" => ChatMessage::user(content),
288 + "assistant" => ChatMessage::assistant(content),
289 + // Tool results flatten to plain text (MVP; acceptable loss).
290 + "tool" => ChatMessage::user(format!("[tool result]\n{content}")),
291 + _ => continue,
292 + };
293 + self.engine.push_history(message).await;
294 + }
295 + }
296 +
297 + async fn compact_history(&self, keep_last: usize) -> Result<(), BackendError> {
298 + let snapshot = self.engine.history().await;
299 + // One plain (tool-free) inference round produces the summary. On error
300 + // history is untouched — send_message only mutates it on success, and
301 + // whatever it appended is wiped by the clear below anyway.
302 + let result = self
303 + .engine
304 + .send_message(SUMMARIZE_PROMPT)
305 + .await
306 + .map_err(|error| error.to_string())?;
307 + // Local models may reason in <think> blocks; keep only the visible part.
308 + let (_think, summary) = crate::chat::strip_think_blocks(&result.text);
309 +
310 + self.engine.clear_history().await;
311 + // Leading system messages carry the session context; keep them all.
312 + for message in snapshot
313 + .iter()
314 + .take_while(|message| message.role == ChatRole::System)
315 + {
316 + self.engine.push_history(message.clone()).await;
317 + }
318 + self.engine
319 + .push_history(ChatMessage::user(format!(
320 + "[Conversation summary]\n{summary}"
321 + )))
322 + .await;
323 + let non_system: Vec<&ChatMessage> = snapshot
324 + .iter()
325 + .filter(|message| message.role != ChatRole::System)
326 + .collect();
327 + let tail_start = non_system.len().saturating_sub(keep_last);
328 + for message in &non_system[tail_start..] {
329 + self.engine.push_history((*message).clone()).await;
330 + }
331 + Ok(())
332 + }
333 }
334
335 /// Drain an onde streaming receiver, forwarding each token to `sink` and
@@ -591,6 +708,68 @@ impl InferenceBackend for OpenAiBackend {
708 fn is_remote(&self) -> bool {
709 true
710 }
711 +
712 + async fn history_snapshot(&self) -> Vec<serde_json::Value> {
713 + self.history.lock().await.clone()
714 + }
715 +
716 + async fn restore_history(&self, history: Vec<serde_json::Value>) {
717 + // The snapshot includes the seeded system message, so a wholesale
718 + // replacement restores exactly what was saved.
719 + *self.history.lock().await = history;
720 + }
721 +
722 + async fn compact_history(&self, keep_last: usize) -> Result<(), BackendError> {
723 + let snapshot: Vec<serde_json::Value> = self.history.lock().await.clone();
724 +
725 + // Ask the endpoint for a summary of the conversation so far, through
726 + // the ordinary completion machinery (non-streaming).
727 + self.history
728 + .lock()
729 + .await
730 + .push(serde_json::json!({ "role": "user", "content": SUMMARIZE_PROMPT }));
731 + let summary = match self.complete(None, None).await {
732 + Ok(result) => result.text,
733 + Err(error) => {
734 + // Roll back the summarization request; the turn never happened.
735 + *self.history.lock().await = snapshot;
736 + return Err(error);
737 + }
738 + };
739 +
740 + let system = snapshot
741 + .first()
742 + .filter(|message| message["role"] == "system")
743 + .cloned();
744 + let non_system: Vec<serde_json::Value> = snapshot
745 + .iter()
746 + .filter(|message| message["role"] != "system")
747 + .cloned()
748 + .collect();
749 + let tail_start = non_system.len().saturating_sub(keep_last);
750 + let mut tail = non_system[tail_start..].to_vec();
751 + // Drop leading tool results whose assistant tool-call message was
752 + // summarized away — strict endpoints reject orphaned `role: "tool"`
753 + // entries on the very next request.
754 + while tail
755 + .first()
756 + .is_some_and(|message| message["role"] == "tool")
757 + {
758 + tail.remove(0);
759 + }
760 +
761 + let mut rebuilt = Vec::new();
762 + if let Some(system) = system {
763 + rebuilt.push(system);
764 + }
765 + rebuilt.push(serde_json::json!({
766 + "role": "user",
767 + "content": format!("[Conversation summary]\n{summary}"),
768 + }));
769 + rebuilt.extend(tail);
770 + *self.history.lock().await = rebuilt;
771 + Ok(())
772 + }
773 }
774
775 // ── OpenAI response shapes ────────────────────────────────────────────────────────
@@ -783,6 +962,159 @@ mod tests {
962 assert_eq!(last["content"], "cancelled by the user");
963 }
964
965 + #[test]
966 + fn estimate_tokens_scales_with_serialized_size() {
967 + assert_eq!(estimate_tokens(&[]), 0);
968 +
969 + let short = vec![serde_json::json!({ "role": "user", "content": "hi" })];
970 + let long = vec![serde_json::json!({ "role": "user", "content": "x".repeat(4_000) })];
971 + let short_estimate = estimate_tokens(&short);
972 + let long_estimate = estimate_tokens(&long);
973 +
974 + assert!(short_estimate > 0, "non-empty history estimates > 0 tokens");
975 + assert!(long_estimate > short_estimate, "longer history costs more");
976 + // 4,000 content chars / 4 ≈ 1,000 tokens, plus a little JSON framing.
977 + assert!((1_000..1_100).contains(&long_estimate), "{long_estimate}");
978 + }
979 +
980 + #[tokio::test]
981 + async fn openai_snapshot_restore_round_trips_exactly() {
982 + let backend = OpenAiBackend::new("http://localhost", "", "m", Some("be helpful".into()));
983 + {
984 + let mut history = backend.history.lock().await;
985 + history.push(serde_json::json!({ "role": "user", "content": "hello" }));
986 + history.push(streamed_assistant_history(
987 + "",
988 + &[ToolCall {
989 + id: "call_1".to_string(),
990 + name: "read_file".to_string(),
991 + arguments: r#"{"path":"a.rs"}"#.to_string(),
992 + }],
993 + ));
994 + history.push(serde_json::json!({
995 + "role": "tool", "tool_call_id": "call_1", "content": "fn main() {}",
996 + }));
997 + history.push(serde_json::json!({ "role": "assistant", "content": "done" }));
998 + }
999 + let snapshot = backend.history_snapshot().await;
1000 + assert_eq!(
1001 + snapshot[0]["role"], "system",
1002 + "snapshot keeps the system message"
1003 + );
1004 +
1005 + // Restoring into a backend seeded with a *different* system prompt must
1006 + // replace everything, including that seed.
1007 + let restored = OpenAiBackend::new("http://localhost", "", "m", Some("other seed".into()));
1008 + restored.restore_history(snapshot.clone()).await;
1009 + assert_eq!(restored.history_snapshot().await, snapshot);
1010 + }
1011 +
1012 + /// Minimal scripted OpenAI-compatible endpoint: accepts one HTTP request on
1013 + /// a std listener and answers with a fixed non-streaming completion.
1014 + fn spawn_completion_stub(summary: &str) -> std::net::SocketAddr {
1015 + use std::io::{Read, Write};
1016 +
1017 + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
1018 + let addr = listener.local_addr().unwrap();
1019 + let body = serde_json::json!({
1020 + "choices": [{ "message": { "role": "assistant", "content": summary } }]
1021 + })
1022 + .to_string();
1023 + std::thread::spawn(move || {
1024 + let (mut stream, _) = listener.accept().unwrap();
1025 + // Read until the full request (headers + content-length body) is in.
1026 + let mut request = Vec::new();
1027 + let mut chunk = [0u8; 4096];
1028 + loop {
1029 + let n = stream.read(&mut chunk).unwrap_or(0);
1030 + if n == 0 {
1031 + break;
1032 + }
1033 + request.extend_from_slice(&chunk[..n]);
1034 + if let Some(headers_end) =
1035 + request.windows(4).position(|window| window == b"\r\n\r\n")
1036 + {
1037 + let headers = String::from_utf8_lossy(&request[..headers_end]);
1038 + let content_length = headers
1039 + .lines()
1040 + .find_map(|line| {
1041 + line.to_ascii_lowercase()
1042 + .strip_prefix("content-length:")
1043 + .map(|value| value.trim().parse::<usize>().unwrap_or(0))
1044 + })
1045 + .unwrap_or(0);
1046 + if request.len() >= headers_end + 4 + content_length {
1047 + break;
1048 + }
1049 + }
1050 + }
1051 + let response = format!(
1052 + "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\n\
1053 + content-length: {}\r\nconnection: close\r\n\r\n{}",
1054 + body.len(),
1055 + body
1056 + );
1057 + let _ = stream.write_all(response.as_bytes());
1058 + });
1059 + addr
1060 + }
1061 +
1062 + #[tokio::test]
1063 + async fn compact_history_rebuilds_system_summary_and_tail() {
1064 + let addr = spawn_completion_stub("We refactored backend.rs; tests pass.");
1065 + let backend = OpenAiBackend::new(
1066 + format!("http://{addr}/v1"),
1067 + "test-key",
1068 + "test-model",
1069 + Some("be helpful".into()),
1070 + );
1071 + {
1072 + let mut history = backend.history.lock().await;
1073 + for i in 0..5 {
1074 + let role = if i % 2 == 0 { "user" } else { "assistant" };
1075 + history.push(serde_json::json!({
1076 + "role": role, "content": format!("message {i}"),
1077 + }));
1078 + }
1079 + }
1080 +
1081 + backend.compact_history(2).await.unwrap();
1082 +
1083 + let history = backend.history_snapshot().await;
1084 + assert_eq!(history.len(), 4, "system + summary + last 2: {history:?}");
1085 + assert_eq!(history[0]["role"], "system");
1086 + assert_eq!(history[0]["content"], "be helpful");
1087 + assert_eq!(history[1]["role"], "user");
1088 + let summary_text = history[1]["content"].as_str().unwrap();
1089 + assert!(summary_text.starts_with("[Conversation summary]\n"));
1090 + assert!(summary_text.contains("We refactored backend.rs; tests pass."));
1091 + assert_eq!(
1092 + history[2],
1093 + serde_json::json!({ "role": "assistant", "content": "message 3" })
1094 + );
1095 + assert_eq!(
1096 + history[3],
1097 + serde_json::json!({ "role": "user", "content": "message 4" })
1098 + );
1099 + }
1100 +
1101 + #[tokio::test]
1102 + async fn compact_history_failure_leaves_history_intact() {
1103 + // No listener at this address: the summarization request fails, and
1104 + // history must roll back to exactly what it was.
1105 + let backend =
1106 + OpenAiBackend::new("http://127.0.0.1:9", "", "test-model", Some("sys".into()));
1107 + backend
1108 + .history
1109 + .lock()
1110 + .await
1111 + .push(serde_json::json!({ "role": "user", "content": "hello" }));
1112 + let before = backend.history_snapshot().await;
1113 +
1114 + assert!(backend.compact_history(2).await.is_err());
1115 + assert_eq!(backend.history_snapshot().await, before);
1116 + }
1117 +
1118 #[test]
1119 fn assistant_message_with_tool_calls_round_trips() {
1120 let message = ResponseMessage {
src/chat.rs
+86 -2
@@ -777,6 +777,10 @@ mod tui {
777 Plan(Option<bool>),
778 /// Show the effective permission policy for this session.
779 Permissions,
780 + /// Summarize-and-shrink the conversation history on demand.
781 + Compact,
782 + /// Restore the saved TUI session from disk.
783 + Resume,
784 Exit,
785 Unknown(String),
786 }
@@ -803,6 +807,8 @@ mod tui {
807 "/whoami" => SlashCommand::Whoami,
808 "/plan" => SlashCommand::Plan(parse_on_off(arg)),
809 "/permissions" => SlashCommand::Permissions,
810 + "/compact" => SlashCommand::Compact,
811 + "/resume" => SlashCommand::Resume,
812 "/exit" | "/quit" | "/q" => SlashCommand::Exit,
813 other => SlashCommand::Unknown(other.to_string()),
814 })
@@ -1407,6 +1413,8 @@ mod tui {
1413 /whoami — show the signed-in account\n\
1414 /plan [on|off] — plan mode: research only, no edits or commands\n\
1415 /permissions — show the tool permission policy\n\
1416 + /compact — summarize and shrink conversation history\n\
1417 + /resume — restore the saved session from disk\n\
1418 /clear — wipe conversation history\n\
1419 /status — show engine status\n\
1420 /exit — quit chat",
@@ -1416,10 +1424,53 @@ mod tui {
1424 let cleared = engine.clear_history().await;
1425 app.messages.clear();
1426 crate::permissions::reset_session(crate::permissions::TUI_SESSION);
1427 + // The saved session must not resurrect what the user just wiped.
1428 + crate::session_store::delete(TUI_STORE_SESSION);
1429 app.messages.push(ChatMessage::system(format!(
1430 "Cleared {cleared} turn(s). History is empty.",
1431 )));
1432 }
1433 + SlashCommand::Compact => {
1434 + let before = crate::backend::estimate_tokens(&app.backend.history_snapshot().await);
1435 + match app
1436 + .backend
1437 + .compact_history(crate::backend::COMPACT_KEEP_LAST)
1438 + .await
1439 + {
1440 + Ok(()) => {
1441 + let snapshot = app.backend.history_snapshot().await;
1442 + let after = crate::backend::estimate_tokens(&snapshot);
1443 + // Keep the saved session in step with the compacted state.
1444 + if let Err(error) = crate::session_store::save(TUI_STORE_SESSION, &snapshot)
1445 + {
1446 + log::warn!("session save after /compact failed: {error}");
1447 + }
1448 + app.messages.push(ChatMessage::system(format!(
1449 + "Compacted history: ~{before} → ~{after} tokens (estimated)."
1450 + )));
1451 + }
1452 + Err(error) => {
1453 + app.messages
1454 + .push(ChatMessage::system(format!("Compaction failed: {error}")));
1455 + }
1456 + }
1457 + }
1458 + SlashCommand::Resume => match crate::session_store::load(TUI_STORE_SESSION) {
1459 + Some(history) if !history.is_empty() => {
1460 + let restored = history.len();
1461 + app.backend.restore_history(history).await;
1462 + app.messages.push(ChatMessage::system(format!(
1463 + "Restored {restored} message(s) from the saved session. \
1464 + The model remembers the conversation; the scrollback above does not \
1465 + replay it."
1466 + )));
1467 + }
1468 + _ => {
1469 + app.messages.push(ChatMessage::system(
1470 + "No saved session to resume. Sessions are saved after each turn.",
1471 + ));
1472 + }
1473 + },
1474 SlashCommand::Plan(value) => {
1475 use crate::permissions::{self, TUI_SESSION};
1476 let enabled = value.unwrap_or_else(|| !permissions::plan_mode(TUI_SESSION));
@@ -1575,8 +1626,13 @@ mod tui {
1626
1627 // ── Background inference task ─────────────────────────────────────────────
1628
1578 - /// cap tool rounds so a confused model can't loop forever
1579 - const MAX_TOOL_ROUNDS: usize = 10;
1629 + /// cap tool rounds so a confused model can't loop forever; auto-compaction
1630 + /// keeps long runs inside the context window, so the cap can be generous
1631 + const MAX_TOOL_ROUNDS: usize = 24;
1632 +
1633 + /// The TUI is a single conversation, so it persists under one fixed
1634 + /// session-store id (ACP sessions use their protocol-assigned ids).
1635 + const TUI_STORE_SESSION: &str = "tui";
1636
1637 fn build_tool_specs() -> Vec<ToolSpec> {
1638 let mut specs: Vec<ToolSpec> = crate::tools::all_tools()
@@ -1686,6 +1742,27 @@ mod tui {
1742 round += 1;
1743 log::info!("tool round {} — {} call(s)", round, result.tool_calls.len());
1744
1745 + // Auto-compaction: long tool runs grow history fast; fold it into
1746 + // a summary before the next round rather than blowing the window.
1747 + let estimate = crate::backend::estimate_tokens(&backend.history_snapshot().await);
1748 + if estimate > crate::backend::DEFAULT_CONTEXT_TOKEN_BUDGET {
1749 + log::info!(
1750 + "history ≈{estimate} tokens exceeds budget {} — compacting",
1751 + crate::backend::DEFAULT_CONTEXT_TOKEN_BUDGET
1752 + );
1753 + match backend
1754 + .compact_history(crate::backend::COMPACT_KEEP_LAST)
1755 + .await
1756 + {
1757 + Ok(()) => {
1758 + let after =
1759 + crate::backend::estimate_tokens(&backend.history_snapshot().await);
1760 + log::info!("compacted history to ≈{after} tokens");
1761 + }
1762 + Err(error) => log::warn!("history compaction failed: {error}"),
1763 + }
1764 + }
1765 +
1766 let mut tool_results = Vec::new();
1767
1768 for (call_index, tc) in result.tool_calls.iter().enumerate() {
@@ -1823,6 +1900,13 @@ mod tui {
1900 }
1901 }
1902
1903 + // Persist the completed turn so /resume (or a restart) can pick the
1904 + // conversation back up.
1905 + let snapshot = backend.history_snapshot().await;
1906 + if let Err(error) = crate::session_store::save(TUI_STORE_SESSION, &snapshot) {
1907 + log::warn!("session save failed: {error}");
1908 + }
1909 +
1910 log::info!("inference complete — {} tool round(s)", round);
1911 // tx drops here — event loop gets None from rx.recv()
1912 }
src/main.rs
+70 -3
@@ -37,6 +37,7 @@ mod mcp;
37 mod models;
38 mod permissions;
39 mod provider;
40 +mod session_store;
41 mod settings;
42 mod setup;
43 mod skills;
@@ -229,8 +230,10 @@ pub(crate) fn system_prompt_for_model(tool_calling: bool) -> &'static str {
230 }
231 }
232
232 -/// cap tool-call loops so a confused model can't spin forever
233 -const MAX_TOOL_ROUNDS: usize = 10;
233 +/// cap tool-call loops so a confused model can't spin forever; auto-compaction
234 +/// (see [`backend::DEFAULT_CONTEXT_TOKEN_BUDGET`]) keeps long runs inside the
235 +/// context window, so the cap can afford to be generous
236 +const MAX_TOOL_ROUNDS: usize = 24;
237
238 /// Outcome of asking the client for permission to run one tool call.
239 enum PermissionVerdict {
@@ -769,6 +772,7 @@ impl SiGitAgent {
772 "on|off (optional)",
773 ),
774 AvailableCommand::new("permissions", "Show the tool permission policy"),
775 + AvailableCommand::new("compact", "Summarize and shrink the conversation history"),
776 AvailableCommand::new("clear", "Wipe the conversation history"),
777 AvailableCommand::new("status", "Show engine status"),
778 ];
@@ -961,7 +965,7 @@ impl SiGitAgent {
965 log::warn!("could not set cwd to {}: {err}", args.cwd.display());
966 }
967
964 - // no session persistence, so "load" just resets
968 + // start from a clean slate; a stored session (below) replaces it
969 self.engine.clear_history().await;
970
971 self.engine
@@ -973,6 +977,20 @@ impl SiGitAgent {
977 // Honor the persisted Local Inference toggle (off + signed in → cloud).
978 self.apply_startup_inference_mode().await;
979
980 + // Durable sessions: when this session id was saved before, restore its
981 + // history into the active backend. The snapshot includes the system
982 + // messages that were live when it was saved, so restore replaces the
983 + // freshly seeded state wholesale.
984 + if let Some(history) = session_store::load(&args.session_id.to_string()) {
985 + let restored = history.len();
986 + let backend = self.backend.lock().await.clone();
987 + backend.restore_history(history).await;
988 + log::info!(
989 + "load_session: restored {restored} message(s) for {}",
990 + args.session_id
991 + );
992 + }
993 +
994 let config_options = {
995 let guard = self.current_model.lock().unwrap();
996 build_model_config_options(&guard)
@@ -1300,6 +1318,25 @@ impl SiGitAgent {
1318 result.tool_calls.len()
1319 );
1320
1321 + // Auto-compaction: long tool runs grow history fast; fold it into
1322 + // a summary before the next round rather than blowing the window.
1323 + let estimate = backend::estimate_tokens(&backend.history_snapshot().await);
1324 + if estimate > backend::DEFAULT_CONTEXT_TOKEN_BUDGET {
1325 + log::info!(
1326 + "prompt({}) history ≈{} tokens exceeds budget {} — compacting",
1327 + session_id,
1328 + estimate,
1329 + backend::DEFAULT_CONTEXT_TOKEN_BUDGET
1330 + );
1331 + match backend.compact_history(backend::COMPACT_KEEP_LAST).await {
1332 + Ok(()) => {
1333 + let after = backend::estimate_tokens(&backend.history_snapshot().await);
1334 + log::info!("prompt({}) compacted to ≈{} tokens", session_id, after);
1335 + }
1336 + Err(error) => log::warn!("prompt({}) compaction failed: {error}", session_id),
1337 + }
1338 + }
1339 +
1340 let mut tool_results = Vec::new();
1341
1342 for (call_index, tc) in result.tool_calls.iter().enumerate() {
@@ -1417,6 +1454,13 @@ impl SiGitAgent {
1454 }
1455 }
1456
1457 + // Persist the completed turn so a restart (or session/load) can pick
1458 + // the conversation back up.
1459 + let snapshot = backend.history_snapshot().await;
1460 + if let Err(error) = session_store::save(&session_id.to_string(), &snapshot) {
1461 + log::warn!("prompt({}) session save failed: {error}", session_id);
1462 + }
1463 +
1464 log::info!("prompt({}) complete — {} tool round(s)", session_id, round);
1465 Ok(PromptResponse::new(StopReason::EndTurn))
1466 }
@@ -2155,6 +2199,8 @@ enum SlashCommand {
2199 Plan(Option<bool>),
2200 /// Show the effective permission policy for this session.
2201 Permissions,
2202 + /// Summarize-and-shrink the conversation history on demand.
2203 + Compact,
2204 Exit,
2205 Unknown(String),
2206 }
@@ -2182,6 +2228,7 @@ fn parse_slash(input: &str) -> Option<SlashCommand> {
2228 "/reload" => SlashCommand::Reload,
2229 "/plan" => SlashCommand::Plan(parse_on_off(argument)),
2230 "/permissions" => SlashCommand::Permissions,
2231 + "/compact" => SlashCommand::Compact,
2232 "/exit" | "/quit" | "/q" => SlashCommand::Exit,
2233 other => SlashCommand::Unknown(other.to_string()),
2234 })
@@ -2292,6 +2339,7 @@ async fn exec_slash_acp(
2339 /reload - re-sync sign-in and model state\n\
2340 /plan [on|off] - plan mode: research only, no edits or commands\n\
2341 /permissions - show the tool permission policy\n\
2342 + /compact - summarize and shrink conversation history\n\
2343 /clear - wipe conversation history\n\
2344 /status - show engine status\n\
2345 /exit - end this turn",
@@ -2301,6 +2349,8 @@ async fn exec_slash_acp(
2349 SlashCommand::Clear => {
2350 let cleared = agent.engine.clear_history().await;
2351 permissions::reset_session(&session_id.to_string());
2352 + // The saved session must not resurrect what the user just wiped.
2353 + session_store::delete(&session_id.to_string());
2354 agent
2355 .send_assistant_message(
2356 cx,
@@ -2326,6 +2376,23 @@ async fn exec_slash_acp(
2376 let summary = permissions::describe(&session_id.to_string());
2377 agent.send_assistant_message(cx, session_id, summary).ok();
2378 }
2379 + SlashCommand::Compact => {
2380 + let backend = agent.backend.lock().await.clone();
2381 + let before = backend::estimate_tokens(&backend.history_snapshot().await);
2382 + let message = match backend.compact_history(backend::COMPACT_KEEP_LAST).await {
2383 + Ok(()) => {
2384 + let snapshot = backend.history_snapshot().await;
2385 + let after = backend::estimate_tokens(&snapshot);
2386 + // Keep the saved session in step with the compacted state.
2387 + if let Err(error) = session_store::save(&session_id.to_string(), &snapshot) {
2388 + log::warn!("session save after /compact failed: {error}");
2389 + }
2390 + format!("Compacted history: ~{before} → ~{after} tokens (estimated).")
2391 + }
2392 + Err(error) => format!("Compaction failed: {error}"),
2393 + };
2394 + agent.send_assistant_message(cx, session_id, message).ok();
2395 + }
2396 SlashCommand::Status => {
2397 let info = agent.engine.info().await;
2398 let model = info.model_name.as_deref().unwrap_or("(none)");
src/session_store.rs new
+167
@@ -0,0 +1,167 @@
1 +//! Durable session storage.
2 +//!
3 +//! One JSON-lines file per session at `$SIGIT_CONFIG_DIR/sessions/<id>.jsonl`
4 +//! (config dir resolution matches [`crate::settings`] / [`crate::credentials`]:
5 +//! `$SIGIT_CONFIG_DIR` or `~/.config/sigit`). Each line is one history message
6 +//! as produced by `InferenceBackend::history_snapshot`, so a saved file can be
7 +//! restored into either backend.
8 +//!
9 +//! Writes are atomic (temp file + rename) so a crash mid-save never leaves a
10 +//! truncated session behind. Session ids are sanitized to a filename-safe
11 +//! alphabet before touching the filesystem.
12 +
13 +use std::path::PathBuf;
14 +
15 +use serde_json::Value;
16 +
17 +/// Config directory: `$SIGIT_CONFIG_DIR` or `~/.config/sigit`.
18 +fn config_dir() -> Option<PathBuf> {
19 + if let Ok(dir) = std::env::var("SIGIT_CONFIG_DIR") {
20 + return Some(PathBuf::from(dir));
21 + }
22 + let home = std::env::var("HOME").ok()?;
23 + Some(PathBuf::from(home).join(".config/sigit"))
24 +}
25 +
26 +fn sessions_dir() -> Option<PathBuf> {
27 + config_dir().map(|dir| dir.join("sessions"))
28 +}
29 +
30 +/// Reduce a session id to a filename-safe form: `[A-Za-z0-9._-]` pass through,
31 +/// anything else becomes `_`. An empty id maps to `_` so the file name never
32 +/// collapses to just the extension.
33 +fn sanitize_id(session_id: &str) -> String {
34 + if session_id.is_empty() {
35 + return "_".to_string();
36 + }
37 + session_id
38 + .chars()
39 + .map(|c| {
40 + if c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-') {
41 + c
42 + } else {
43 + '_'
44 + }
45 + })
46 + .collect()
47 +}
48 +
49 +fn session_path(session_id: &str) -> Option<PathBuf> {
50 + sessions_dir().map(|dir| dir.join(format!("{}.jsonl", sanitize_id(session_id))))
51 +}
52 +
53 +/// Persist a history snapshot for `session_id`, replacing any previous save.
54 +/// The write is atomic: a temp file in the same directory is renamed over the
55 +/// final path.
56 +pub fn save(session_id: &str, history: &[Value]) -> Result<(), String> {
57 + let path =
58 + session_path(session_id).ok_or_else(|| "cannot resolve config directory".to_string())?;
59 + let dir = path
60 + .parent()
61 + .ok_or_else(|| "session path has no parent".to_string())?;
62 + std::fs::create_dir_all(dir).map_err(|error| format!("create {dir:?}: {error}"))?;
63 +
64 + let mut body = String::new();
65 + for message in history {
66 + body.push_str(&message.to_string());
67 + body.push('\n');
68 + }
69 +
70 + // Unique temp name so two processes saving the same session can't clobber
71 + // each other's half-written file; rename is atomic on the same filesystem.
72 + let tmp = dir.join(format!(
73 + ".{}.{}.tmp",
74 + sanitize_id(session_id),
75 + std::process::id()
76 + ));
77 + std::fs::write(&tmp, body).map_err(|error| format!("write {tmp:?}: {error}"))?;
78 + std::fs::rename(&tmp, &path).map_err(|error| {
79 + let _ = std::fs::remove_file(&tmp);
80 + format!("rename {tmp:?} -> {path:?}: {error}")
81 + })?;
82 + Ok(())
83 +}
84 +
85 +/// Load the saved history for `session_id`, or `None` when no save exists (or
86 +/// it cannot be read). Unparseable lines are skipped rather than failing the
87 +/// whole restore.
88 +pub fn load(session_id: &str) -> Option<Vec<Value>> {
89 + let path = session_path(session_id)?;
90 + let contents = std::fs::read_to_string(&path).ok()?;
91 + Some(
92 + contents
93 + .lines()
94 + .filter(|line| !line.trim().is_empty())
95 + .filter_map(|line| serde_json::from_str::<Value>(line).ok())
96 + .collect(),
97 + )
98 +}
99 +
100 +/// Remove the saved history for `session_id`. Missing files are fine.
101 +pub fn delete(session_id: &str) {
102 + if let Some(path) = session_path(session_id) {
103 + let _ = std::fs::remove_file(path);
104 + }
105 +}
106 +
107 +#[cfg(test)]
108 +mod tests {
109 + use super::*;
110 +
111 + #[test]
112 + fn sanitize_keeps_safe_chars_and_replaces_the_rest() {
113 + assert_eq!(sanitize_id("abc-DEF_123.z"), "abc-DEF_123.z");
114 + assert_eq!(sanitize_id("a/b\\c:d e"), "a_b_c_d_e");
115 + assert_eq!(sanitize_id("../../etc/passwd"), ".._.._etc_passwd");
116 + assert_eq!(sanitize_id(""), "_");
117 + }
118 +
119 + // One test for the filesystem behavior because it mutates the
120 + // process-global `SIGIT_CONFIG_DIR` env var (same pattern as the settings
121 + // tests): splitting would race under the parallel test runner.
122 + #[test]
123 + fn save_load_delete_round_trip() {
124 + let _guard = crate::ENV_TEST_LOCK
125 + .lock()
126 + .unwrap_or_else(|poisoned| poisoned.into_inner());
127 + let dir = std::env::temp_dir().join(format!("sigit_sessions_{}", std::process::id()));
128 + let _ = std::fs::remove_dir_all(&dir);
129 + // SAFETY: serialized by ENV_TEST_LOCK; restored below.
130 + unsafe { std::env::set_var("SIGIT_CONFIG_DIR", &dir) };
131 +
132 + // Missing file → None.
133 + assert_eq!(load("nope"), None);
134 +
135 + let history = vec![
136 + serde_json::json!({ "role": "system", "content": "sys" }),
137 + serde_json::json!({ "role": "user", "content": "hi\nthere" }),
138 + serde_json::json!({
139 + "role": "assistant", "content": null,
140 + "tool_calls": [{ "id": "call_1", "type": "function",
141 + "function": { "name": "read_file", "arguments": "{}" } }],
142 + }),
143 + ];
144 + save("sess-1", &history).unwrap();
145 + assert_eq!(load("sess-1"), Some(history.clone()));
146 +
147 + // Saving again replaces, not appends.
148 + let shorter = vec![serde_json::json!({ "role": "user", "content": "only" })];
149 + save("sess-1", &shorter).unwrap();
150 + assert_eq!(load("sess-1"), Some(shorter));
151 +
152 + // A hostile id stays inside the sessions dir via sanitization.
153 + save("../escape", &history).unwrap();
154 + assert!(dir.join("sessions").join(".._escape.jsonl").is_file());
155 + assert_eq!(load("../escape"), Some(history));
156 + delete("../escape");
157 + assert_eq!(load("../escape"), None);
158 +
159 + delete("sess-1");
160 + assert_eq!(load("sess-1"), None);
161 + // Deleting a missing session is a no-op.
162 + delete("sess-1");
163 +
164 + unsafe { std::env::remove_var("SIGIT_CONFIG_DIR") };
165 + let _ = std::fs::remove_dir_all(&dir);
166 + }
167 +}