384
self.messages.push(ChatMessage::system(
385
"In this world, nothing can be said to be certain, except death and taxes. ~ Pak Sigit",
386
));
387
- self.messages.push(ChatMessage::system(format!(
388
- "Current model: {}",
389
- self.current_model_name
390
- )));
387
+ if self.backend.is_remote() {
388
+ self.messages.push(ChatMessage::system(format!(
389
+ "Current model: {}",
390
+ self.current_model_name
391
+ )));
392
+ } else {
393
+ // On-device models are never loaded implicitly; prompt the user to
394
+ // load one explicitly before their first message.
395
+ self.messages.push(ChatMessage::system(format!(
396
+ "No on-device model loaded. Run /load to load {}, or /models to choose one.",
397
+ self.current_model_name
398
+ )));
399
+ }
400
self.messages
401
.push(ChatMessage::system("Type /help for commands."));
402
}
662
Status,
663
/// picker UI, or jump straight to model N
664
Models(Option<usize>),
665
+ /// explicitly load the selected (or default) on-device model
666
+ Load,
667
/// `/login <email> <password>` — the raw argument, parsed when executed.
668
Login(Option<String>),
669
Logout,
685
"/clear" => SlashCommand::Clear,
686
"/status" => SlashCommand::Status,
687
"/models" => SlashCommand::Models(arg.and_then(|s| s.parse::<usize>().ok())),
688
+ "/load" => SlashCommand::Load,
689
"/login" => SlashCommand::Login(arg.map(str::to_string)),
690
"/logout" => SlashCommand::Logout,
691
"/whoami" => SlashCommand::Whoami,
1141
}
1142
}
1143
1144
+ // ── Explicit on-device model loading ──────────────────────────────────────
1145
+
1146
+ /// The local model `/load` should bring up: the persisted selection if it
1147
+ /// still resolves to a known model, otherwise the first on-device (non-cloud)
1148
+ /// entry in the picker.
1149
+ fn default_local_model_item(app: &App) -> Option<ModelPickerItem> {
1150
+ if let Some(selected) = crate::setup::load_selected_model()
1151
+ && let Some(item) = app.model_picker_items.iter().find(|item| {
1152
+ item.config.model_id == selected.model_id
1153
+ && item
1154
+ .config
1155
+ .files
1156
+ .iter()
1157
+ .any(|file| file == &selected.gguf_file)
1158
+ })
1159
+ {
1160
+ return Some(item.clone());
1161
+ }
1162
+ app.model_picker_items
1163
+ .iter()
1164
+ .find(|item| item.cloud_tier.is_none())
1165
+ .cloned()
1166
+ }
1167
+
1168
+ /// Load `model` on-device on a dedicated loader thread, routing inference to a
1169
+ /// fresh `LocalBackend` and driving the switch-progress UI. The caller is
1170
+ /// responsible for any cloud-tier handling; this path is on-device only.
1171
+ fn start_local_model_load<B: ratatui::backend::Backend>(
1172
+ app: &mut App,
1173
+ model: ModelPickerItem,
1174
+ engine: Arc<ChatEngine>,
1175
+ terminal: &mut ratatui::Terminal<B>,
1176
+ ) {
1177
+ if model.cache_health == ModelCacheHealth::Incomplete {
1178
+ app.messages.push(ChatMessage::system(format!(
1179
+ "error: {} has an incomplete local cache and cannot be selected yet.",
1180
+ model.display_name
1181
+ )));
1182
+ return;
1183
+ }
1184
+
1185
+ // Route inference on-device; the loader thread below fills the engine the
1186
+ // LocalBackend reads from.
1187
+ app.backend = Arc::new(LocalBackend::new(Arc::clone(&engine)));
1188
+
1189
+ let loading_msg = if model.cache_health == ModelCacheHealth::NotDownloaded {
1190
+ format!(
1191
+ "Downloading and loading {} ({})… this may take a few minutes.",
1192
+ model.display_name, model.description
1193
+ )
1194
+ } else {
1195
+ format!("Loading {}…", model.display_name)
1196
+ };
1197
+
1198
+ app.messages.push(ChatMessage::system(loading_msg));
1199
+ terminal.draw(|frame| render(frame, app)).ok();
1200
+
1201
+ let (tx, rx) = mpsc::channel(1);
1202
+ app.model_load_rx = Some(rx);
1203
+ app.switching_model = true;
1204
+ app.switching_model_id = Some(model.config.model_id.clone());
1205
+ // Only show download progress for models not yet cached.
1206
+ app.download_progress = if model.cache_health == ModelCacheHealth::NotDownloaded {
1207
+ Some((0, 0))
1208
+ } else {
1209
+ None
1210
+ };
1211
+
1212
+ let sampling = SamplingConfig {
1213
+ max_tokens: Some(model.max_tokens),
1214
+ ..SamplingConfig::default()
1215
+ };
1216
+
1217
+ // own thread + runtime so block_in_place doesn't starve the TUI loop
1218
+ let system_prompt = crate::system_prompt_for_model(model.tool_calling);
1219
+ let engine_handle = Arc::clone(&engine);
1220
+ let tool_calling = model.tool_calling;
1221
+ std::thread::spawn(move || {
1222
+ let rt = tokio::runtime::Runtime::new().expect("failed to create model-loader runtime");
1223
+ let update = rt.block_on(async move {
1224
+ match engine_handle
1225
+ .load_gguf_model(
1226
+ model.config.clone(),
1227
+ Some(system_prompt.to_string()),
1228
+ Some(sampling),
1229
+ )
1230
+ .await
1231
+ {
1232
+ Ok(_) => ModelLoadUpdate::Loaded(model.display_name.clone()),
1233
+ Err(err) => ModelLoadUpdate::Error(err.to_string()),
1234
+ }
1235
+ });
1236
+ // capacity-1 channel, receiver alive while switching
1237
+ let _ = tx.blocking_send(update);
1238
+ });
1239
+ // applied on ModelLoadUpdate::Loaded
1240
+ app.pending_tool_calling = Some(tool_calling);
1241
+ }
1242
+
1243
// ── Slash command execution ───────────────────────────────────────────────
1244
1245
async fn exec_slash<B: ratatui::backend::Backend>(
1254
"/help — show this message\n\
1255
/models — open the model picker\n\
1256
/models N — switch to model N\n\
1257
+ /load — load the selected on-device model\n\
1258
/login E P — sign in to siGit Code Cloud\n\
1259
/logout — sign out\n\
1260
/whoami — show the signed-in account\n\
1322
return;
1323
}
1324
1213
- if model.cache_health == ModelCacheHealth::Incomplete {
1214
- app.close_model_picker();
1215
- app.messages.push(ChatMessage::system(format!(
1216
- "error: {} has an incomplete local cache and cannot be selected yet.",
1217
- model.display_name
1218
- )));
1219
- return;
1220
- }
1221
-
1222
- // Route inference on-device; the loader thread below
1223
- // fills the engine the LocalBackend reads from.
1224
- app.backend = Arc::new(LocalBackend::new(Arc::clone(&engine)));
1225
-
1226
- let loading_msg = if model.cache_health
1227
- == ModelCacheHealth::NotDownloaded
1228
- {
1229
- format!(
1230
- "Downloading and loading {} ({})… this may take a few minutes.",
1231
- model.display_name, model.description
1232
- )
1233
- } else {
1234
- format!("Loading {}…", model.display_name)
1235
- };
1236
-
1325
app.close_model_picker();
1238
- app.messages.push(ChatMessage::system(loading_msg));
1239
- terminal.draw(|frame| render(frame, app)).ok();
1240
-
1241
- let (tx, rx) = mpsc::channel(1);
1242
- app.model_load_rx = Some(rx);
1243
- app.switching_model = true;
1244
- app.switching_model_id = Some(model.config.model_id.clone());
1245
- // Only show download progress for models not yet cached.
1246
- app.download_progress =
1247
- if model.cache_health == ModelCacheHealth::NotDownloaded {
1248
- Some((0, 0))
1249
- } else {
1250
- None
1251
- };
1252
-
1253
- let sampling = SamplingConfig {
1254
- max_tokens: Some(model.max_tokens),
1255
- ..SamplingConfig::default()
1256
- };
1257
-
1258
- // own thread + runtime so block_in_place doesn't starve the TUI loop
1259
- let system_prompt = crate::system_prompt_for_model(model.tool_calling);
1260
- let engine_handle = Arc::clone(&engine);
1261
- let tool_calling = model.tool_calling;
1262
- std::thread::spawn(move || {
1263
- let rt = tokio::runtime::Runtime::new()
1264
- .expect("failed to create model-loader runtime");
1265
- let update = rt.block_on(async move {
1266
- match engine_handle
1267
- .load_gguf_model(
1268
- model.config.clone(),
1269
- Some(system_prompt.to_string()),
1270
- Some(sampling),
1271
- )
1272
- .await
1273
- {
1274
- Ok(_) => {
1275
- ModelLoadUpdate::Loaded(model.display_name.clone())
1276
- }
1277
- Err(err) => ModelLoadUpdate::Error(err.to_string()),
1278
- }
1279
- });
1280
- // capacity-1 channel, receiver alive while switching
1281
- let _ = tx.blocking_send(update);
1282
- });
1283
- // applied on ModelLoadUpdate::Loaded
1284
- app.pending_tool_calling = Some(tool_calling);
1326
+ start_local_model_load(app, model, Arc::clone(&engine), terminal);
1327
}
1328
}
1329
}
1330
},
1331
+ SlashCommand::Load => match default_local_model_item(app) {
1332
+ None => {
1333
+ app.messages.push(ChatMessage::system(
1334
+ "No local model available to load. Use /models to see the list.",
1335
+ ));
1336
+ }
1337
+ Some(model) => {
1338
+ start_local_model_load(app, model, Arc::clone(&engine), terminal);
1339
+ }
1340
+ },
1341
SlashCommand::Login(arg) => {
1342
let message = match arg.as_deref().and_then(crate::account::parse_login_args) {
1343
Some((email, password)) => {
1755
continue;
1756
}
1757
1758
+ // On-device inference needs a model in memory, and we
1759
+ // never load one implicitly: the user loads it with
1760
+ // /load (or /models). Refuse rather than erroring out
1761
+ // deep in the backend.
1762
+ if !app.backend.is_remote()
1763
+ && engine.info().await.status == onde::inference::EngineStatus::Unloaded
1764
+ {
1765
+ app.messages.push(ChatMessage::user(&text));
1766
+ app.messages.push(ChatMessage::system(
1767
+ "No on-device model is loaded. Run /load to load the selected \
1768
+ model, or /models to choose one.",
1769
+ ));
1770
+ continue;
1771
+ }
1772
+
1773
// ── spawn inference ──────────────────────────────
1774
app.messages.push(ChatMessage::user(&text));
1775
app.start_thinking();