34
mod credentials;
35
mod models;
36
mod provider;
37
+mod settings;
38
mod setup;
39
mod tools;
40
41
+/// Serializes tests that mutate process-global env vars (`SIGIT_CONFIG_DIR`
42
+/// etc.). `cargo test` runs tests in parallel within a binary, so without this
43
+/// the credentials and settings round-trip tests clobber each other's env.
44
+#[cfg(test)]
45
+pub(crate) static ENV_TEST_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
46
+
47
use std::io::IsTerminal;
48
#[cfg(unix)]
49
use std::io::{BufWriter, Write};
667
"model number to switch to (optional)",
668
)),
669
),
670
+ with_hint(
671
+ "local",
672
+ "Toggle on-device inference mode",
673
+ "on|off (optional)",
674
+ ),
675
with_hint("login", "Sign in to siGit Code Cloud", "<email> <password>"),
676
AvailableCommand::new("logout", "Sign out of siGit Code Cloud"),
677
AvailableCommand::new("whoami", "Show the signed-in account"),
880
)))
881
.await;
882
883
+ // Honor the persisted Local Inference toggle (off + signed in → cloud).
884
+ self.apply_startup_inference_mode().await;
885
+
886
let config_options = {
887
let guard = self.current_model.lock().unwrap();
888
build_model_config_options(&guard)
931
)))
932
.await;
933
934
+ // Honor the persisted Local Inference toggle (off + signed in → cloud).
935
+ self.apply_startup_inference_mode().await;
936
+
937
let config_options = {
938
let guard = self.current_model.lock().unwrap();
939
build_model_config_options(&guard)
980
)))
981
.await;
982
983
+ // Honor the persisted Local Inference toggle (off + signed in → cloud).
984
+ self.apply_startup_inference_mode().await;
985
+
986
let config_options = {
987
let guard = self.current_model.lock().unwrap();
988
build_model_config_options(&guard)
1320
*guard = cloud_config;
1321
}
1322
1323
+ // Explicitly choosing a cloud tier puts us in cloud mode.
1324
+ let _ = settings::set_local_inference(false);
1325
+
1326
log::info!("switched to cloud tier {tier}");
1327
Some(cfg.display_name)
1328
}
1329
1330
+ /// Apply the persisted Local Inference mode at session start. When local
1331
+ /// inference is off and an account is signed in, route to a cloud tier so the
1332
+ /// on-device model is never loaded; otherwise leave the on-device backend in
1333
+ /// place. Call after the session cwd is set so the cloud system prompt picks
1334
+ /// it up. Does not flip the stored setting on the not-signed-in fallback.
1335
+ async fn apply_startup_inference_mode(&self) {
1336
+ if settings::local_inference_enabled() {
1337
+ return;
1338
+ }
1339
+ if self.switch_to_cloud_tier("balanced").await.is_some() {
1340
+ log::info!("startup: local inference off; routing inference to siGit Code Cloud");
1341
+ } else {
1342
+ log::warn!(
1343
+ "local inference is off but no account is signed in; staying on-device. \
1344
+ Run /login or set Local Inference on."
1345
+ );
1346
+ }
1347
+ }
1348
+
1349
/// Route inference back on-device. Used after leaving a cloud tier for a
1350
/// local model. The `LocalBackend` reads the live `engine`, so this just
1351
/// repoints the active backend.
1434
args.value
1435
);
1436
1437
+ // ── Local Inference toggle ──────────────────────────────────────────
1438
+ if args.config_id.0.as_ref() == LOCAL_INFERENCE_CONFIG_ID {
1439
+ let enabled = match args.value.0.as_ref() {
1440
+ LOCAL_INFERENCE_ON => true,
1441
+ LOCAL_INFERENCE_OFF => false,
1442
+ other => {
1443
+ return Err(agent_client_protocol::Error::new(
1444
+ -32602,
1445
+ format!("unknown Local Inference value: {other}"),
1446
+ ));
1447
+ }
1448
+ };
1449
+ if let Err(error) = settings::set_local_inference(enabled) {
1450
+ return Err(agent_client_protocol::Error::new(
1451
+ -32603,
1452
+ format!("could not save Local Inference setting: {error}"),
1453
+ ));
1454
+ }
1455
+ let message = if enabled {
1456
+ "Local inference is on. On-device models are highlighted; pick one from Model."
1457
+ } else {
1458
+ "Local inference is off. siGit Code Cloud tiers are highlighted; pick one from Model."
1459
+ };
1460
+ self.send_assistant_message(cx, args.session_id.clone(), format!("\n\n{message}"))
1461
+ .ok();
1462
+ // Rebuild so the Model picker reflects the new emphasis/order.
1463
+ let current = self.current_model.lock().unwrap().clone();
1464
+ let config_options = build_model_config_options(¤t);
1465
+ return Ok(SetSessionConfigOptionResponse::new(config_options));
1466
+ }
1467
+
1468
if args.config_id.0.as_ref() != MODEL_CONFIG_ID {
1469
return Err(agent_client_protocol::Error::new(
1470
-32602,
1719
Ok(new_config) => {
1720
// Route inference back on-device (in case we were on a cloud tier).
1721
self.reset_to_local_backend().await;
1722
+ // Selecting an on-device model puts us in local mode.
1723
+ let _ = settings::set_local_inference(true);
1724
1725
let completion_title = if needs_download {
1726
format!("✓ {} downloaded and loaded", new_config.display_name)
1779
/// config option ID for the model picker in Zed's agent panel
1780
const MODEL_CONFIG_ID: &str = "sigit-model";
1781
1782
+/// config option ID for the Local Inference on/off toggle. Surfaced as a
1783
+/// two-option `select` so ACP clients without slash-command support (e.g. Xcode)
1784
+/// can still flip the mode from the agent panel.
1785
+const LOCAL_INFERENCE_CONFIG_ID: &str = "sigit-local-inference";
1786
+
1787
+/// `select` value ids for the Local Inference toggle.
1788
+const LOCAL_INFERENCE_ON: &str = "local-inference-on";
1789
+const LOCAL_INFERENCE_OFF: &str = "local-inference-off";
1790
+
1791
/// Replace non-ASCII chars so a downstream byte-index truncation can't split a
1792
/// multi-byte char. Zed slices the model-picker label at a fixed byte offset
1793
/// (`agent_ui/src/config_options.rs`) and panics — crashing the whole editor —
1803
// The full list, including the siGit Code Cloud tiers, so the panel picker
1804
// mirrors the TUI `/models`. Cloud entries are sign-in gated at selection.
1805
let items = models::build_model_picker_items();
1806
+ let active_kind = models::active_inference_kind();
1807
1808
let options: Vec<SessionConfigSelectOption> = items
1809
.iter()
1810
.filter(|item| item.cache_health != setup::ModelCacheHealth::Incomplete)
1811
.map(|item| {
1812
let mut desc_parts = Vec::new();
1813
+ // Mark options in the inactive mode so the active group reads as the
1814
+ // recommended set (the list is already ordered active-group-first).
1815
+ if item.source.kind() != active_kind {
1816
+ desc_parts.push("inactive mode".to_string());
1817
+ }
1818
if item.tool_calling {
1819
desc_parts.push("tool calling".to_string());
1820
}
1855
})
1856
.collect();
1857
1858
+ // Local Inference on/off toggle, modeled as a two-option select so panel-only
1859
+ // ACP clients (no slash commands) can flip the mode.
1860
+ let local_on = settings::local_inference_enabled();
1861
+ let local_current = SessionConfigValueId::new(if local_on {
1862
+ LOCAL_INFERENCE_ON
1863
+ } else {
1864
+ LOCAL_INFERENCE_OFF
1865
+ });
1866
+ let local_options = vec![
1867
+ SessionConfigSelectOption::new(
1868
+ SessionConfigValueId::new(LOCAL_INFERENCE_ON),
1869
+ "On (on-device)".to_string(),
1870
+ )
1871
+ .description("Run inference on-device; on-device models are highlighted".to_string()),
1872
+ SessionConfigSelectOption::new(
1873
+ SessionConfigValueId::new(LOCAL_INFERENCE_OFF),
1874
+ "Off (siGit Code Cloud)".to_string(),
1875
+ )
1876
+ .description("Use siGit Code Cloud; cloud tiers are highlighted".to_string()),
1877
+ ];
1878
+ let local_option = SessionConfigOption::select(
1879
+ LOCAL_INFERENCE_CONFIG_ID,
1880
+ "Local Inference",
1881
+ local_current,
1882
+ local_options,
1883
+ )
1884
+ .description("Toggle on-device inference; changes which models are highlighted");
1885
+
1886
if options.is_empty() {
1768
- return vec![];
1887
+ return vec![local_option];
1888
}
1889
1890
let current_value = SessionConfigValueId::new(current_model.model_id.as_str());
1893
SessionConfigOption::select(MODEL_CONFIG_ID, "Model", current_value, options)
1894
.category(SessionConfigOptionCategory::Model)
1895
.description("Select an on-device model or a siGit Code Cloud tier"),
1896
+ local_option,
1897
]
1898
}
1899
1917
Clear,
1918
Status,
1919
Models(Option<usize>),
1920
+ /// toggle on-device inference mode. `Some(true/false)` sets it, `None` flips it.
1921
+ Local(Option<bool>),
1922
/// `/login <email> <password>` — the raw argument, parsed when executed.
1923
Login(Option<String>),
1924
Logout,
1942
"/clear" => SlashCommand::Clear,
1943
"/status" => SlashCommand::Status,
1944
"/models" => SlashCommand::Models(argument.and_then(|v| v.parse::<usize>().ok())),
1945
+ "/local" => SlashCommand::Local(parse_on_off(argument)),
1946
"/login" => SlashCommand::Login(argument.map(str::to_string)),
1947
"/logout" => SlashCommand::Logout,
1948
"/whoami" => SlashCommand::Whoami,
1952
})
1953
}
1954
1955
+/// `on`/`off` (and synonyms) → `Some(bool)`; missing or unrecognized → `None`
1956
+/// (meaning "toggle the current value").
1957
+fn parse_on_off(arg: Option<&str>) -> Option<bool> {
1958
+ match arg.map(|s| s.trim().to_ascii_lowercase())?.as_str() {
1959
+ "on" | "true" | "1" | "yes" => Some(true),
1960
+ "off" | "false" | "0" | "no" => Some(false),
1961
+ _ => None,
1962
+ }
1963
+}
1964
+
1965
fn format_models_list(current_model: &GgufModelConfig) -> String {
1966
let items = models::build_model_picker_items();
1967
if items.is_empty() {
2047
"/help - show this message\n\
2048
/models - list available models\n\
2049
/models N - switch to model N\n\
2050
+ /local [on|off]- toggle on-device inference mode\n\
2051
/login E P - sign in to siGit Code Cloud\n\
2052
/logout - sign out\n\
2053
/whoami - show the signed-in account\n\
2138
match agent.switch_model_by_id(&model.config.model_id).await {
2139
Ok(new_config) => {
2140
agent.reset_to_local_backend().await;
2141
+ let _ = settings::set_local_inference(true);
2142
agent.engine.clear_history().await;
2143
agent
2144
.send_assistant_message(
2172
2173
let switched = agent.switch_model_by_id(&model.config.model_id).await?;
2174
agent.reset_to_local_backend().await;
2175
+ let _ = settings::set_local_inference(true);
2176
agent.engine.clear_history().await;
2177
2178
agent
2186
}
2187
}
2188
}
2189
+ SlashCommand::Local(value) => {
2190
+ let enabled = value.unwrap_or(!settings::local_inference_enabled());
2191
+ let message = match settings::set_local_inference(enabled) {
2192
+ Ok(()) if enabled => "Local inference is on. On-device models are highlighted; \
2193
+ pick one with /models."
2194
+ .to_string(),
2195
+ Ok(()) => "Local inference is off. siGit Code Cloud tiers are highlighted; \
2196
+ pick one with /models."
2197
+ .to_string(),
2198
+ Err(error) => format!("error: could not save local inference setting: {error}"),
2199
+ };
2200
+ agent
2201
+ .send_assistant_message(cx, session_id.clone(), message)
2202
+ .ok();
2203
+ // Refresh the panel so the Model picker reflects the new emphasis.
2204
+ let config_options = {
2205
+ let current = agent.current_model.lock().unwrap();
2206
+ build_model_config_options(¤t)
2207
+ };
2208
+ agent
2209
+ .send_tool_call_update(
2210
+ cx,
2211
+ session_id,
2212
+ SessionUpdate::ConfigOptionUpdate(ConfigOptionUpdate::new(config_options)),
2213
+ )
2214
+ .ok();
2215
+ }
2216
SlashCommand::Login(argument) => {
2217
let message = match argument.as_deref().and_then(account::parse_login_args) {
2218
Some((email, password)) => match account::authenticate(&email, &password).await {
2440
(backend, label)
2441
}
2442
None => {
2280
- // On-device: load the local GGUF model on a real thread.
2281
- let loader_engine = Arc::clone(&engine);
2282
- let system_prompt = system_prompt_for_model(tool_calling).to_string();
2283
- std::thread::spawn(move || {
2284
- let rt =
2285
- tokio::runtime::Runtime::new().expect("failed to create loader runtime");
2286
- let result = rt.block_on(loader_engine.load_gguf_model(
2287
- config,
2288
- Some(system_prompt),
2289
- Some(sampling),
2290
- ));
2291
- let _ = load_tx.send(result.map(|_| ()).map_err(|e| e.to_string()));
2292
- });
2293
- let backend =
2294
- Arc::new(LocalBackend::new(Arc::clone(&engine))) as Arc<dyn InferenceBackend>;
2295
- (backend, startup_model_name)
2443
+ // Honor the Local Inference toggle: when off and signed in, start
2444
+ // on a cloud tier instead of loading an on-device model. When off
2445
+ // but not signed in, fall back to on-device (a usable backend) —
2446
+ // the user can /login or /local on.
2447
+ let cloud_when_off = if settings::local_inference_enabled() {
2448
+ None
2449
+ } else {
2450
+ provider::cloud_tier_provider("balanced")
2451
+ };
2452
+
2453
+ match cloud_when_off {
2454
+ Some(provider) => {
2455
+ log::info!(
2456
+ "inference: local inference off; using {} (model {})",
2457
+ provider.display_name,
2458
+ provider.model
2459
+ );
2460
+ let _ = load_tx.send(Ok(()));
2461
+ let label = provider.display_name.clone();
2462
+ let backend = Arc::new(OpenAiBackend::new(
2463
+ provider.base_url,
2464
+ provider.api_key,
2465
+ provider.model,
2466
+ Some(SYSTEM_PROMPT.to_string()),
2467
+ )) as Arc<dyn InferenceBackend>;
2468
+ (backend, label)
2469
+ }
2470
+ None => {
2471
+ if !settings::local_inference_enabled() {
2472
+ log::warn!(
2473
+ "local inference is off but no account is signed in; \
2474
+ falling back to on-device. Run /login or /local on."
2475
+ );
2476
+ }
2477
+ // On-device: load the local GGUF model on a real thread.
2478
+ let loader_engine = Arc::clone(&engine);
2479
+ let system_prompt = system_prompt_for_model(tool_calling).to_string();
2480
+ std::thread::spawn(move || {
2481
+ let rt = tokio::runtime::Runtime::new()
2482
+ .expect("failed to create loader runtime");
2483
+ let result = rt.block_on(loader_engine.load_gguf_model(
2484
+ config,
2485
+ Some(system_prompt),
2486
+ Some(sampling),
2487
+ ));
2488
+ let _ = load_tx.send(result.map(|_| ()).map_err(|e| e.to_string()));
2489
+ });
2490
+ let backend = Arc::new(LocalBackend::new(Arc::clone(&engine)))
2491
+ as Arc<dyn InferenceBackend>;
2492
+ (backend, startup_model_name)
2493
+ }
2494
+ }
2495
}
2496
};
2497