@setoelkahfi / sigit / commits / 339b4ef

Load on-device model only on explicit request

Stop loading the on-device GGUF model implicitly. The TUI and ACP sessions now come up immediately and the local model is brought into memory only when the user runs the new /load command or actively picks a model in /models. Prompts sent before a model is loaded return a hint instead of blocking on a multi-minute download. The goal is no automatic load on first launch or when a new thread opens: a fresh `sigit` TUI run, and ACP new/load/fork sessions plus Zed re-firing the last model selection, all leave the engine unloaded. - TUI: skip the startup loader thread; add /load; gate prompt submission and the welcome banner on whether a local model is loaded - ACP: drop the lazy load on first prompt; add /load (advertised + /help); refuse local prompts when the engine is unloaded - ACP: a re-fired panel selection stays a no-op so opening a thread never auto-loads; only actively picking a different model loads - Factor default_local_model_config() shared by startup and /load Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

paydii committed Jun 28, 2026 at 15:08 UTC 339b4ef65c08c319fb728c401dba19786a37ab4b
3 files changed +215 -136
CHANGELOG.md
+6
@@ -1,5 +1,11 @@
1 # Changelog
2
3 +## Unreleased
4 +
5 +### What changed
6 +
7 +- On-device models are no longer loaded implicitly. The chat UI and ACP sessions come up immediately, and the local model is brought into memory only when you run the new `/load` command (or pick one in `/models`). Prompts sent before a model is loaded now return a hint instead of blocking on a multi-minute download.
8 +
9 ## 1.2.2
10
11 Streams assistant tokens as they arrive, on-device and over the cloud.
src/chat.rs
+142 -75
@@ -384,10 +384,19 @@ mod tui {
384 self.messages.push(ChatMessage::system(
385 "In this world, nothing can be said to be certain, except death and taxes. ~ Pak Sigit",
386 ));
387 - self.messages.push(ChatMessage::system(format!(
388 - "Current model: {}",
389 - self.current_model_name
390 - )));
387 + if self.backend.is_remote() {
388 + self.messages.push(ChatMessage::system(format!(
389 + "Current model: {}",
390 + self.current_model_name
391 + )));
392 + } else {
393 + // On-device models are never loaded implicitly; prompt the user to
394 + // load one explicitly before their first message.
395 + self.messages.push(ChatMessage::system(format!(
396 + "No on-device model loaded. Run /load to load {}, or /models to choose one.",
397 + self.current_model_name
398 + )));
399 + }
400 self.messages
401 .push(ChatMessage::system("Type /help for commands."));
402 }
@@ -653,6 +662,8 @@ mod tui {
662 Status,
663 /// picker UI, or jump straight to model N
664 Models(Option<usize>),
665 + /// explicitly load the selected (or default) on-device model
666 + Load,
667 /// `/login <email> <password>` — the raw argument, parsed when executed.
668 Login(Option<String>),
669 Logout,
@@ -674,6 +685,7 @@ mod tui {
685 "/clear" => SlashCommand::Clear,
686 "/status" => SlashCommand::Status,
687 "/models" => SlashCommand::Models(arg.and_then(|s| s.parse::<usize>().ok())),
688 + "/load" => SlashCommand::Load,
689 "/login" => SlashCommand::Login(arg.map(str::to_string)),
690 "/logout" => SlashCommand::Logout,
691 "/whoami" => SlashCommand::Whoami,
@@ -1129,6 +1141,105 @@ mod tui {
1141 }
1142 }
1143
1144 + // ── Explicit on-device model loading ──────────────────────────────────────
1145 +
1146 + /// The local model `/load` should bring up: the persisted selection if it
1147 + /// still resolves to a known model, otherwise the first on-device (non-cloud)
1148 + /// entry in the picker.
1149 + fn default_local_model_item(app: &App) -> Option<ModelPickerItem> {
1150 + if let Some(selected) = crate::setup::load_selected_model()
1151 + && let Some(item) = app.model_picker_items.iter().find(|item| {
1152 + item.config.model_id == selected.model_id
1153 + && item
1154 + .config
1155 + .files
1156 + .iter()
1157 + .any(|file| file == &selected.gguf_file)
1158 + })
1159 + {
1160 + return Some(item.clone());
1161 + }
1162 + app.model_picker_items
1163 + .iter()
1164 + .find(|item| item.cloud_tier.is_none())
1165 + .cloned()
1166 + }
1167 +
1168 + /// Load `model` on-device on a dedicated loader thread, routing inference to a
1169 + /// fresh `LocalBackend` and driving the switch-progress UI. The caller is
1170 + /// responsible for any cloud-tier handling; this path is on-device only.
1171 + fn start_local_model_load<B: ratatui::backend::Backend>(
1172 + app: &mut App,
1173 + model: ModelPickerItem,
1174 + engine: Arc<ChatEngine>,
1175 + terminal: &mut ratatui::Terminal<B>,
1176 + ) {
1177 + if model.cache_health == ModelCacheHealth::Incomplete {
1178 + app.messages.push(ChatMessage::system(format!(
1179 + "error: {} has an incomplete local cache and cannot be selected yet.",
1180 + model.display_name
1181 + )));
1182 + return;
1183 + }
1184 +
1185 + // Route inference on-device; the loader thread below fills the engine the
1186 + // LocalBackend reads from.
1187 + app.backend = Arc::new(LocalBackend::new(Arc::clone(&engine)));
1188 +
1189 + let loading_msg = if model.cache_health == ModelCacheHealth::NotDownloaded {
1190 + format!(
1191 + "Downloading and loading {} ({})… this may take a few minutes.",
1192 + model.display_name, model.description
1193 + )
1194 + } else {
1195 + format!("Loading {}…", model.display_name)
1196 + };
1197 +
1198 + app.messages.push(ChatMessage::system(loading_msg));
1199 + terminal.draw(|frame| render(frame, app)).ok();
1200 +
1201 + let (tx, rx) = mpsc::channel(1);
1202 + app.model_load_rx = Some(rx);
1203 + app.switching_model = true;
1204 + app.switching_model_id = Some(model.config.model_id.clone());
1205 + // Only show download progress for models not yet cached.
1206 + app.download_progress = if model.cache_health == ModelCacheHealth::NotDownloaded {
1207 + Some((0, 0))
1208 + } else {
1209 + None
1210 + };
1211 +
1212 + let sampling = SamplingConfig {
1213 + max_tokens: Some(model.max_tokens),
1214 + ..SamplingConfig::default()
1215 + };
1216 +
1217 + // own thread + runtime so block_in_place doesn't starve the TUI loop
1218 + let system_prompt = crate::system_prompt_for_model(model.tool_calling);
1219 + let engine_handle = Arc::clone(&engine);
1220 + let tool_calling = model.tool_calling;
1221 + std::thread::spawn(move || {
1222 + let rt = tokio::runtime::Runtime::new().expect("failed to create model-loader runtime");
1223 + let update = rt.block_on(async move {
1224 + match engine_handle
1225 + .load_gguf_model(
1226 + model.config.clone(),
1227 + Some(system_prompt.to_string()),
1228 + Some(sampling),
1229 + )
1230 + .await
1231 + {
1232 + Ok(_) => ModelLoadUpdate::Loaded(model.display_name.clone()),
1233 + Err(err) => ModelLoadUpdate::Error(err.to_string()),
1234 + }
1235 + });
1236 + // capacity-1 channel, receiver alive while switching
1237 + let _ = tx.blocking_send(update);
1238 + });
1239 + // applied on ModelLoadUpdate::Loaded
1240 + app.pending_tool_calling = Some(tool_calling);
1241 + }
1242 +
1243 // ── Slash command execution ───────────────────────────────────────────────
1244
1245 async fn exec_slash<B: ratatui::backend::Backend>(
@@ -1143,6 +1254,7 @@ mod tui {
1254 "/help — show this message\n\
1255 /models — open the model picker\n\
1256 /models N — switch to model N\n\
1257 + /load — load the selected on-device model\n\
1258 /login E P — sign in to siGit Code Cloud\n\
1259 /logout — sign out\n\
1260 /whoami — show the signed-in account\n\
@@ -1210,82 +1322,22 @@ mod tui {
1322 return;
1323 }
1324
1213 - if model.cache_health == ModelCacheHealth::Incomplete {
1214 - app.close_model_picker();
1215 - app.messages.push(ChatMessage::system(format!(
1216 - "error: {} has an incomplete local cache and cannot be selected yet.",
1217 - model.display_name
1218 - )));
1219 - return;
1220 - }
1221 -
1222 - // Route inference on-device; the loader thread below
1223 - // fills the engine the LocalBackend reads from.
1224 - app.backend = Arc::new(LocalBackend::new(Arc::clone(&engine)));
1225 -
1226 - let loading_msg = if model.cache_health
1227 - == ModelCacheHealth::NotDownloaded
1228 - {
1229 - format!(
1230 - "Downloading and loading {} ({})… this may take a few minutes.",
1231 - model.display_name, model.description
1232 - )
1233 - } else {
1234 - format!("Loading {}…", model.display_name)
1235 - };
1236 -
1325 app.close_model_picker();
1238 - app.messages.push(ChatMessage::system(loading_msg));
1239 - terminal.draw(|frame| render(frame, app)).ok();
1240 -
1241 - let (tx, rx) = mpsc::channel(1);
1242 - app.model_load_rx = Some(rx);
1243 - app.switching_model = true;
1244 - app.switching_model_id = Some(model.config.model_id.clone());
1245 - // Only show download progress for models not yet cached.
1246 - app.download_progress =
1247 - if model.cache_health == ModelCacheHealth::NotDownloaded {
1248 - Some((0, 0))
1249 - } else {
1250 - None
1251 - };
1252 -
1253 - let sampling = SamplingConfig {
1254 - max_tokens: Some(model.max_tokens),
1255 - ..SamplingConfig::default()
1256 - };
1257 -
1258 - // own thread + runtime so block_in_place doesn't starve the TUI loop
1259 - let system_prompt = crate::system_prompt_for_model(model.tool_calling);
1260 - let engine_handle = Arc::clone(&engine);
1261 - let tool_calling = model.tool_calling;
1262 - std::thread::spawn(move || {
1263 - let rt = tokio::runtime::Runtime::new()
1264 - .expect("failed to create model-loader runtime");
1265 - let update = rt.block_on(async move {
1266 - match engine_handle
1267 - .load_gguf_model(
1268 - model.config.clone(),
1269 - Some(system_prompt.to_string()),
1270 - Some(sampling),
1271 - )
1272 - .await
1273 - {
1274 - Ok(_) => {
1275 - ModelLoadUpdate::Loaded(model.display_name.clone())
1276 - }
1277 - Err(err) => ModelLoadUpdate::Error(err.to_string()),
1278 - }
1279 - });
1280 - // capacity-1 channel, receiver alive while switching
1281 - let _ = tx.blocking_send(update);
1282 - });
1283 - // applied on ModelLoadUpdate::Loaded
1284 - app.pending_tool_calling = Some(tool_calling);
1326 + start_local_model_load(app, model, Arc::clone(&engine), terminal);
1327 }
1328 }
1329 }
1330 },
1331 + SlashCommand::Load => match default_local_model_item(app) {
1332 + None => {
1333 + app.messages.push(ChatMessage::system(
1334 + "No local model available to load. Use /models to see the list.",
1335 + ));
1336 + }
1337 + Some(model) => {
1338 + start_local_model_load(app, model, Arc::clone(&engine), terminal);
1339 + }
1340 + },
1341 SlashCommand::Login(arg) => {
1342 let message = match arg.as_deref().and_then(crate::account::parse_login_args) {
1343 Some((email, password)) => {
@@ -1703,6 +1755,21 @@ mod tui {
1755 continue;
1756 }
1757
1758 + // On-device inference needs a model in memory, and we
1759 + // never load one implicitly: the user loads it with
1760 + // /load (or /models). Refuse rather than erroring out
1761 + // deep in the backend.
1762 + if !app.backend.is_remote()
1763 + && engine.info().await.status == onde::inference::EngineStatus::Unloaded
1764 + {
1765 + app.messages.push(ChatMessage::user(&text));
1766 + app.messages.push(ChatMessage::system(
1767 + "No on-device model is loaded. Run /load to load the selected \
1768 + model, or /models to choose one.",
1769 + ));
1770 + continue;
1771 + }
1772 +
1773 // ── spawn inference ──────────────────────────────
1774 app.messages.push(ChatMessage::user(&text));
1775 app.start_thinking();
src/main.rs
+67 -61
@@ -660,6 +660,7 @@ impl SiGitAgent {
660 "model number to switch to (optional)",
661 )),
662 ),
663 + AvailableCommand::new("load", "Load the selected on-device model"),
664 with_hint("login", "Sign in to siGit Code Cloud", "<email> <password>"),
665 AvailableCommand::new("logout", "Sign out of siGit Code Cloud"),
666 AvailableCommand::new("whoami", "Show the signed-in account"),
@@ -1131,10 +1132,21 @@ impl SiGitAgent {
1132 let backend = self.backend.lock().await.clone();
1133
1134 // Only on-device inference needs a local model in memory. Cloud tiers run
1134 - // over the network, so skip the lazy load and the readiness wait for them.
1135 - if !backend.is_remote() {
1136 - self.start_startup_model_load_if_needed();
1137 - self.await_model_ready(cx, &session_id).await?;
1135 + // over the network, so they never need a local model. We never load the
1136 + // on-device model implicitly: the user loads it explicitly with `/load`
1137 + // (or by picking one in `/models`). If a prompt arrives before that, guide
1138 + // them rather than blocking on a multi-minute download/load.
1139 + if !backend.is_remote()
1140 + && self.engine.info().await.status == onde::inference::EngineStatus::Unloaded
1141 + {
1142 + self.send_assistant_message(
1143 + cx,
1144 + session_id,
1145 + "No on-device model is loaded. Run `/load` to load the selected model, \
1146 + or `/models` to choose one.",
1147 + )
1148 + .ok();
1149 + return Ok(PromptResponse::new(StopReason::EndTurn));
1150 }
1151
1152 // ── tool-calling loop ────────────────────────────────────────────
@@ -1411,12 +1423,16 @@ impl SiGitAgent {
1423 }
1424 }
1425
1414 - // Zed re-fires the last selection on connect; no-op if it's already loaded
1426 + // Zed re-fires the last selection when a thread opens. That re-fire must
1427 + // not load anything: on-device models are loaded only on an explicit
1428 + // request (`/load`, or actively picking a *different* model below), so a
1429 + // re-fire of the already-current selection is a no-op. Otherwise opening a
1430 + // new thread would silently load the local model — exactly what we avoid.
1431 {
1432 let current = self.current_model.lock().unwrap();
1433 if current.model_id == model_id {
1434 log::info!(
1419 - "set_session_config_option: {} is already the active model, skipping",
1435 + "set_session_config_option: {} is already the active selection, skipping",
1436 current.display_name
1437 );
1438 let config_options = build_model_config_options(&current);
@@ -1797,6 +1813,8 @@ enum SlashCommand {
1813 Clear,
1814 Status,
1815 Models(Option<usize>),
1816 + /// Explicitly load the selected (or default) on-device model.
1817 + Load,
1818 /// `/login <email> <password>` — the raw argument, parsed when executed.
1819 Login(Option<String>),
1820 Logout,
@@ -1820,6 +1838,7 @@ fn parse_slash(input: &str) -> Option<SlashCommand> {
1838 "/clear" => SlashCommand::Clear,
1839 "/status" => SlashCommand::Status,
1840 "/models" => SlashCommand::Models(argument.and_then(|v| v.parse::<usize>().ok())),
1841 + "/load" => SlashCommand::Load,
1842 "/login" => SlashCommand::Login(argument.map(str::to_string)),
1843 "/logout" => SlashCommand::Logout,
1844 "/whoami" => SlashCommand::Whoami,
@@ -1914,6 +1933,7 @@ async fn exec_slash_acp(
1933 "/help - show this message\n\
1934 /models - list available models\n\
1935 /models N - switch to model N\n\
1936 + /load - load the selected on-device model\n\
1937 /login E P - sign in to siGit Code Cloud\n\
1938 /logout - sign out\n\
1939 /whoami - show the signed-in account\n\
@@ -2050,6 +2070,25 @@ async fn exec_slash_acp(
2070 }
2071 }
2072 }
2073 + SlashCommand::Load => {
2074 + // Explicitly load the on-device model. This is the only path that
2075 + // brings a local model into memory; prompts never do it implicitly.
2076 + // If a cloud tier is active, fall back to a local default so we don't
2077 + // try to load the (file-less) cloud config as GGUF.
2078 + let on_cloud = {
2079 + let guard = agent.current_model.lock().unwrap();
2080 + guard.model_id.starts_with("sigit-cloud:")
2081 + };
2082 + if on_cloud {
2083 + let default_config = default_local_model_config();
2084 + *agent.current_model.lock().unwrap() = default_config;
2085 + agent.reset_to_local_backend().await;
2086 + }
2087 + // `await_model_ready` drives the download/load progress UI and reports
2088 + // success or failure to the editor.
2089 + agent.start_startup_model_load_if_needed();
2090 + agent.await_model_ready(cx, &session_id).await?;
2091 + }
2092 SlashCommand::Login(argument) => {
2093 let message = match argument.as_deref().and_then(account::parse_login_args) {
2094 Some((email, password)) => match account::authenticate(&email, &password).await {
@@ -2219,42 +2258,11 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2258 .map(|selection| selection.display_name.clone())
2259 .unwrap_or_else(|| GgufModelConfig::qwen25_3b().display_name);
2260
2222 - let config = startup_selection
2223 - .as_ref()
2224 - .and_then(|selection| {
2225 - models::local_picker_items()
2226 - .into_iter()
2227 - .find(|item| {
2228 - selection
2229 - .selected_model
2230 - .as_ref()
2231 - .map(|selected| {
2232 - item.config.model_id == selected.model_id
2233 - && item
2234 - .config
2235 - .files
2236 - .iter()
2237 - .any(|file| file == &selected.gguf_file)
2238 - })
2239 - .unwrap_or(false)
2240 - })
2241 - .map(|item| item.config)
2242 - })
2243 - .unwrap_or_else(GgufModelConfig::qwen25_3b);
2244 - let sampling = SamplingConfig {
2245 - max_tokens: Some(8192),
2246 - ..SamplingConfig::default()
2247 - };
2248 -
2249 - // std::sync::mpsc on a real thread so model loading can't starve the TUI
2261 + // Signals the loading phase to finish. On-device models are no longer loaded
2262 + // at startup, so this resolves immediately for both backends; it stays a
2263 + // channel so the loading-phase plumbing in `chat::run_with` is unchanged.
2264 let (load_tx, load_rx) = std::sync::mpsc::channel::<Result<(), String>>();
2265
2252 - let tool_calling = models::local_picker_items()
2253 - .iter()
2254 - .find(|item| item.config.model_id == config.model_id)
2255 - .map(|item| item.tool_calling)
2256 - .unwrap_or(false);
2257 -
2266 // Pick the inference backend: a configured provider if present, else on-device.
2267 let (inference_backend, startup_model_name): (Arc<dyn InferenceBackend>, String) =
2268 match provider::active_provider() {
@@ -2277,19 +2285,10 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2285 (backend, label)
2286 }
2287 None => {
2280 - // On-device: load the local GGUF model on a real thread.
2281 - let loader_engine = Arc::clone(&engine);
2282 - let system_prompt = system_prompt_for_model(tool_calling).to_string();
2283 - std::thread::spawn(move || {
2284 - let rt =
2285 - tokio::runtime::Runtime::new().expect("failed to create loader runtime");
2286 - let result = rt.block_on(loader_engine.load_gguf_model(
2287 - config,
2288 - Some(system_prompt),
2289 - Some(sampling),
2290 - ));
2291 - let _ = load_tx.send(result.map(|_| ()).map_err(|e| e.to_string()));
2292 - });
2288 + // On-device: do NOT load the local GGUF model implicitly. The user
2289 + // loads it explicitly with /load (or /models) from the chat, so the
2290 + // UI comes up immediately without a multi-minute download/load.
2291 + let _ = load_tx.send(Ok(()));
2292 let backend =
2293 Arc::new(LocalBackend::new(Arc::clone(&engine))) as Arc<dyn InferenceBackend>;
2294 (backend, startup_model_name)
@@ -2332,11 +2331,11 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2331
2332 // ── ACP server mode ───────────────────────────────────────────────────────────
2333
2335 -async fn run_acp_server() -> anyhow::Result<()> {
2336 - log::info!("ACP mode — starting agent server");
2337 -
2338 - let startup_selection = setup::startup_model_selection();
2339 - let config = startup_selection
2334 +/// The on-device model `/load` should bring up by default: the persisted
2335 +/// selection if it still resolves to a known local model, otherwise the built-in
2336 +/// default (`qwen25_3b`).
2337 +fn default_local_model_config() -> GgufModelConfig {
2338 + setup::startup_model_selection()
2339 .as_ref()
2340 .and_then(|selection| {
2341 selection.selected_model.as_ref().and_then(|selected| {
@@ -2353,7 +2352,13 @@ async fn run_acp_server() -> anyhow::Result<()> {
2352 .map(|item| item.config)
2353 })
2354 })
2356 - .unwrap_or_else(GgufModelConfig::qwen25_3b);
2355 + .unwrap_or_else(GgufModelConfig::qwen25_3b)
2356 +}
2357 +
2358 +async fn run_acp_server() -> anyhow::Result<()> {
2359 + log::info!("ACP mode — starting agent server");
2360 +
2361 + let config = default_local_model_config();
2362
2363 let needs_download = models::local_picker_items()
2364 .iter()
@@ -2373,8 +2378,9 @@ async fn run_acp_server() -> anyhow::Result<()> {
2378
2379 let engine = Arc::new(ChatEngine::new());
2380
2376 - // Delay model loading until the first real prompt so initialize/session/new
2377 - // stay lightweight and registry auth checks don't trip over model startup.
2381 + // The on-device model is never loaded implicitly; the user loads it with
2382 + // `/load` (or by picking one in `/models`). So initialize/session/new stay
2383 + // lightweight and `model_ready` starts true (nothing is loading).
2384 let model_ready = Arc::new(AtomicBool::new(true));
2385 let startup_model_load_started = Arc::new(AtomicBool::new(false));
2386 let model_load_error: Arc<std::sync::Mutex<Option<String>>> =