diff --git a/README.md b/README.md index b07342f..32aaa37 100644 --- a/README.md +++ b/README.md @@ -80,10 +80,11 @@ Common in-chat commands: - `/branch` or `/restore`: open the branch/restore menu. - `/login`: configure or update provider login settings. - `/logout`: remove saved provider config and associated model entries. +- `/fast`, `/fast on`, `/fast off`, `/fast status`: prefer faster inference when the active provider/model supports it. v0.3.2 supports this for ChatGPT Codex models marked fast-capable. - `/model `: switch to a model from `~/.cass/models.json`. - `/new`: create a new chat for the current directory. - `/resume `: resume a saved chat for the current directory. -- `/status`: show chat id, model, mode, cwd, record count, and current status. +- `/status`: show chat id, model, fast-mode state, mode, cwd, record count, and current status. Helpful keys: diff --git a/docs/README.md b/docs/README.md index 7b82ee7..2db3dbd 100644 --- a/docs/README.md +++ b/docs/README.md @@ -8,7 +8,7 @@ Cassady tools may list, search, and read this directory. Mutating tools are bloc - [Commands](commands.md): CLI forms, global flags, `cass update`, in-chat commands, and keys. - [Configuration](configuration.md): `~/.cass` files, setup, precedence, schema examples, and validation. -- [Providers and models](providers.md): built-in provider presets, custom OpenAI-compatible endpoints, ChatGPT Codex auth, model discovery, and reasoning metadata. +- [Providers and models](providers.md): built-in provider presets, custom OpenAI-compatible endpoints, ChatGPT Codex auth, model discovery, reasoning metadata, and fast-mode support. - [Access modes and tool safety](access-modes.md): what tools can read, write, edit, and run in each mode. - [Experimental Rust embedding API](embedding.md): import Cassady from Rust, start headless sessions, stream events, and handle approvals. - [Workflows](workflows.md): common ways to inspect code, apply edits, run checks, switch models, and resume chats. diff --git a/docs/commands.md b/docs/commands.md index 0e8f568..e040c09 100644 --- a/docs/commands.md +++ b/docs/commands.md @@ -107,12 +107,13 @@ The updater does not invoke `sudo` or administrator prompts. If the install dire Type `/` to open command autocomplete. - `/branch` or `/restore`: open the branch/restore menu for the current conversation family. +- `/fast`, `/fast on`, `/fast off`, `/fast status`: toggle or inspect a persisted fast-mode preference. Fast mode is active only when the current provider/model supports it; v0.3.2 supports ChatGPT Codex models marked fast-capable. - `/login`: configure or update provider login settings, then reload active provider/model config. - `/logout`: remove saved providers and their associated models, then reload active provider/model config when any remain. - `/model `: switch the model for future turns. Autocomplete lists models from `~/.cass/models.json`. - `/new`: create a new chat for the current directory. - `/resume `: resume a saved chat from the current directory. Autocomplete lists matching chats. -- `/status`: show chat id, state, model, access mode, cwd, record count, and current status. +- `/status`: show chat id, state, model, fast-mode state, access mode, cwd, record count, and current status. Local commands can be used only when the agent is idle. diff --git a/docs/configuration.md b/docs/configuration.md index b120bf9..05ac471 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -42,6 +42,7 @@ Example: "default_provider": "openai", "default_model": "gpt-4.1", "default_reasoning_effort": "medium", + "default_fast_mode": false, "default_access_mode": "read-only", "context_message_limit": 80, "model_tool_result_limit": 24000, @@ -56,6 +57,7 @@ Fields: - `default_provider`: optional provider id from `providers.json`. If omitted, Cassady infers the provider from `default_model` when possible. - `default_model`: optional model id to use by default. - `default_reasoning_effort`: optional `off`, `low`, `medium`, or `high`, clamped to model metadata. +- `default_fast_mode`: optional boolean, defaults to `false`. When `true`, Cassady requests faster inference only for provider/model combinations that advertise fast-mode support. - `default_access_mode`: `"read-only"`, `"workspace-edit"`, or `"full-access"`. - `context_message_limit`: optional legacy upper bound for recent non-system messages. Cassady primarily budgets context from model metadata and trims along valid tool-call boundaries. - `model_tool_result_limit`: optional max bytes of tool output sent back to the model. @@ -136,6 +138,9 @@ Example: "required": false, "default_effort": "medium", "request_format": "reasoning_effort" + }, + "fast_mode": { + "supported": false } } ] @@ -156,9 +161,13 @@ Fields: - `required`: optional boolean, defaults to `false`. - `default_effort`: optional `off`, `low`, `medium`, or `high`; defaults to `medium`. Cannot effectively be `off` when `required` is `true`. - `request_format`: optional `reasoning_effort` or `reasoning_object`; defaults to `reasoning_effort`. +- `fast_mode`: optional object. Defaults to unsupported. + - `supported`: optional boolean, defaults to `false`. Setup marks ChatGPT Codex model entries as supported; custom and OpenAI-compatible model entries default to unsupported. Reasoning effort is a runtime per-turn setting. Press `Tab` to cycle it while idle. Provider-streamed reasoning is persisted and sent back in future model context using the provider's reasoning field, such as `reasoning_content` or `reasoning`. +Fast mode is a persisted preference, not a guarantee. Use `/fast` to toggle it while idle. `/status` shows `enabled` only when the preference is on and the current provider/model can honor it; otherwise it reports `off` or `preferred, unavailable ...`. In v0.3.2, fast-mode request shaping is implemented only for `chatgpt-codex`. + ## Precedence - CLI access-mode flags override `default_access_mode` for the current session. diff --git a/docs/glossary.md b/docs/glossary.md index 254e6a4..d7315db 100644 --- a/docs/glossary.md +++ b/docs/glossary.md @@ -14,9 +14,11 @@ **Exact edit**: An `edit` tool replacement where each `old_text` must match exactly once in the original file before anything is written. +**Fast mode**: A saved preference enabled with `/fast`. It is active only when the current provider/model advertises fast-mode support; otherwise Cassady keeps the preference but reports it as unavailable. + **Global instructions**: Optional text in `~/.cass/global.md` included in new chat system prompts. Cassady follows these instructions when they fit the active request, but they cannot override runtime safety constraints such as access modes, tool denials, approvals, or workspace boundaries. -**Model metadata**: The `models.json` entry describing a model id, owning provider, display name, context limits, tool/streaming support, and reasoning behavior. +**Model metadata**: The `models.json` entry describing a model id, owning provider, display name, context limits, tool/streaming support, reasoning behavior, and fast-mode support. **OpenAI-compatible provider**: A provider exposing an API compatible with the OpenAI-style chat/completions behavior Cassady uses. diff --git a/docs/providers.md b/docs/providers.md index 8452bc1..0816750 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -74,7 +74,8 @@ Provider protocols that are not OpenAI-compatible are supported only when Cassad - display name; - context length and max output tokens; - tool and streaming support; -- reasoning support and request format. +- reasoning support and request format; +- fast-mode support. `config.json` selects active defaults, such as `default_provider`, `default_model`, and `default_access_mode`. @@ -90,6 +91,15 @@ Reasoning metadata controls how the runtime reasoning effort behaves: Reasoning display is separate. `show_reasoning` controls whether provider-streamed reasoning is visible in the transcript; press `Ctrl-Shift-R` or `Ctrl-R` to toggle it at runtime. +## Fast-mode metadata + +Fast mode has two parts: + +- `default_fast_mode` in `config.json`: the user's saved preference. +- `fast_mode.supported` in `models.json`: whether the active provider/model can honor that preference. + +In v0.3.2, Cassady sends fast-mode requests only for `ChatGPT Codex`. Setup marks ChatGPT Codex model entries as fast-capable. OpenAI-compatible and custom model entries default to unsupported, so `/fast` can remember the preference without sending provider-specific fields. + ## Switching models Use one of these approaches: @@ -104,7 +114,7 @@ or inside a chat: /model MODEL ``` -The in-chat model autocomplete lists entries from `~/.cass/models.json`. Switching the model also updates the default model and reasoning effort in `config.json` for future sessions. +The in-chat model autocomplete lists entries from `~/.cass/models.json`. Switching the model also updates the default provider, default model, and reasoning effort in `config.json` for future sessions. If fast mode is preferred, Cassady recomputes whether it is active after the switch. ## Health checks diff --git a/docs/workflows.md b/docs/workflows.md index fca6e25..87b1302 100644 --- a/docs/workflows.md +++ b/docs/workflows.md @@ -95,7 +95,7 @@ Inside a chat: /model MODEL_ID ``` -Autocomplete lists models from `~/.cass/models.json`. Switching models is allowed only when idle. Cassady persists the last used model and reasoning effort into `config.json`. +Autocomplete lists models from `~/.cass/models.json`. Switching models is allowed only when idle. Cassady persists the last used provider, model, and reasoning effort into `config.json`. You can also launch with a model override: @@ -103,6 +103,18 @@ You can also launch with a model override: cass --model MODEL_ID ``` +## Prefer fast mode + +Inside a chat: + +```text +/fast +``` + +Use `/fast on`, `/fast off`, or `/fast status` when you want an explicit action. Cassady saves the preference in `config.json`, but fast mode is active only when the current provider/model supports it. In v0.3.2, support is implemented for ChatGPT Codex model entries marked with `fast_mode.supported`. + +Switching to an unsupported provider/model keeps the preference but makes `/status` show fast mode as unavailable. Switching back to a supported ChatGPT Codex model enables it again. + ## Resume a chat List chats for the current directory: diff --git a/src/agent.rs b/src/agent.rs index 762eb33..24b7159 100644 --- a/src/agent.rs +++ b/src/agent.rs @@ -3,7 +3,7 @@ use crate::config::{Config, ReasoningEffort}; use crate::conversation::{now_ts, Conversation, Record, StoredToolCall}; use crate::prompt; use crate::providers::types::ModelMessage; -use crate::providers::ProviderClient; +use crate::providers::{ProviderClient, ProviderRuntimeOptions}; use crate::security::PolicyDecision; use crate::tools::{self, ToolContext, ToolRuntimeEvent}; use anyhow::Result; @@ -88,7 +88,13 @@ pub async fn run_turn_with_commands( let reasoning_effort = settings .reasoning_effort .clamp_for_model(settings.config.model_metadata.as_ref()); - let provider = match ProviderClient::from_config(&settings.config, reasoning_effort) { + let provider = match ProviderClient::from_config( + &settings.config, + ProviderRuntimeOptions { + reasoning_effort, + fast_mode: settings.config.fast_mode_state().active, + }, + ) { Ok(provider) => provider, Err(err) => { append_visible_assistant( diff --git a/src/app.rs b/src/app.rs index 3ccb8d9..c739d9b 100644 --- a/src/app.rs +++ b/src/app.rs @@ -1,6 +1,6 @@ use crate::agent::{self, AgentCommand, AgentEvent, AgentSettings}; use crate::cli::{self, Cli, Command}; -use crate::config::{self, Config, ModelDefinition, ReasoningEffort}; +use crate::config::{self, Config, FastModeState, ModelDefinition, ReasoningEffort}; use crate::conversation::{self, Conversation, Record}; use crate::prompt; use crate::ui::autofill::{AutoFillItem, AutoFillMenu}; @@ -397,6 +397,7 @@ async fn run_tui( show_full_tools, show_reasoning, reasoning_effort, + fast_mode_active: config.fast_mode_state().active, scroll, autofill: if branch_menu.is_some() { None @@ -862,10 +863,46 @@ async fn run_tui( } } } + Ok(LocalCommand::Fast(command)) => { + if busy { + status = "fast mode can be changed when idle".into(); + } else { + input.clear(); + autofill_selected = 0; + match apply_fast_mode_command(&mut config, command) { + Ok(message) => { + transcript.push(TranscriptBlock { + kind: TranscriptKind::Status, + title: "fast".into(), + content: message.clone(), + }); + status = message; + } + Err(err) => { + status = + format!("fast mode update failed: {err}"); + transcript.push(TranscriptBlock { + kind: TranscriptKind::Error, + title: "fast".into(), + content: err.to_string(), + }); + } + } + if stick_to_bottom { + scroll = bottom_scroll( + &terminal, + &input, + &transcript, + show_full_tools, + show_reasoning, + )?; + } + } + } Ok(LocalCommand::Status) => { let content = chat_status( &chat_id, - &config.model, + &config, mode, &cwd, busy, @@ -894,14 +931,13 @@ async fn run_tui( if busy { status = "model can be changed when idle".into(); } else { - config.model = model.clone(); - config.model_metadata = - model_metadata_for(&config, &model)?; + apply_model_selection(&mut config, &model)?; reasoning_effort = ReasoningEffort::default_for_model( config.model_metadata.as_ref(), ); - let _ = crate::config::save_last_used( + let _ = crate::config::save_last_used_provider( &config.root, + &config.provider_id, &config.model, reasoning_effort, ); @@ -910,9 +946,9 @@ async fn run_tui( transcript.push(TranscriptBlock { kind: TranscriptKind::Status, title: "model".into(), - content: format!("model changed to {model}"), + content: model_status_message(&config), }); - status = format!("model: {model}"); + status = model_status_message(&config); if stick_to_bottom { scroll = bottom_scroll( &terminal, @@ -1885,6 +1921,7 @@ fn assistant_content_matches(a: &str, b: &str) -> bool { #[derive(Debug, Clone, PartialEq, Eq)] enum LocalCommand { Branch, + Fast(FastModeCommand), Login, Logout, Model(String), @@ -1893,6 +1930,14 @@ enum LocalCommand { Status, } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum FastModeCommand { + Toggle, + On, + Off, + Status, +} + struct CommandSpec { name: &'static str, usage: &'static str, @@ -1907,6 +1952,12 @@ const COMMANDS: &[CommandSpec] = &[ description: "open branch/restore menu", takes_value: false, }, + CommandSpec { + name: "fast", + usage: "/fast [on|off|status]", + description: "toggle faster Codex inference when supported", + takes_value: false, + }, CommandSpec { name: "login", usage: "/login", @@ -2038,14 +2089,38 @@ fn model_autofill(input: &str, selected: usize, config: &Config) -> Result Result> { +fn apply_model_selection(config: &mut Config, model_id: &str) -> Result<()> { let models = crate::config::load_or_create_default_model_registry(&config.root)?; - Ok(models + let metadata = models .models .iter() .find(|model| model.id == model_id && model.provider == config.provider_id) .cloned() - .or_else(|| models.models.into_iter().find(|model| model.id == model_id))) + .or_else(|| { + models + .models + .iter() + .find(|model| model.id == model_id) + .cloned() + }); + + if let Some(model) = &metadata { + if model.provider != config.provider_id { + let providers = crate::config::load_or_create_default_provider_registry(&config.root)?; + if let Some(provider) = providers + .providers + .iter() + .find(|provider| provider.id == model.provider) + { + config.provider_id = provider.id.clone(); + config.active_provider = provider.to_resolved(); + } + } + } + + config.model = model_id.to_string(); + config.model_metadata = metadata; + Ok(()) } fn model_matches(model: &ModelDefinition, query: &str) -> bool { @@ -2175,6 +2250,19 @@ fn parse_local_command(input: &str) -> std::result::Result } Ok(LocalCommand::Branch) } + "/fast" => { + let command = match parts.next() { + None => FastModeCommand::Toggle, + Some("on") => FastModeCommand::On, + Some("off") => FastModeCommand::Off, + Some("status") => FastModeCommand::Status, + Some(_) => return Err("usage: /fast [on|off|status]".into()), + }; + if parts.next().is_some() { + return Err("usage: /fast [on|off|status]".into()); + } + Ok(LocalCommand::Fast(command)) + } "/login" => { if parts.next().is_some() { return Err("usage: /login".into()); @@ -2223,7 +2311,7 @@ fn parse_local_command(input: &str) -> std::result::Result fn chat_status( chat_id: &str, - model: &str, + config: &Config, mode: crate::access::AccessMode, cwd: &Path, busy: bool, @@ -2231,13 +2319,76 @@ fn chat_status( record_count: usize, ) -> String { format!( - "chat: {chat_id}\nstate: {}\nmodel: {model}\nmode: {mode}\ncwd: {}\nrecords: {record_count}\nstatus: {}", + "chat: {chat_id}\nstate: {}\nmodel: {}\nfast: {}\nmode: {mode}\ncwd: {}\nrecords: {record_count}\nstatus: {}", if busy { "running" } else { "idle" }, + config.model, + fast_mode_status(&config.fast_mode_state()), cwd.display(), if status.is_empty() { "idle" } else { status } ) } +fn apply_fast_mode_command(config: &mut Config, command: FastModeCommand) -> Result { + match command { + FastModeCommand::Status => Ok(fast_mode_status(&config.fast_mode_state())), + FastModeCommand::Toggle | FastModeCommand::On | FastModeCommand::Off => { + let enabled = match command { + FastModeCommand::Toggle => !config.default_fast_mode, + FastModeCommand::On => true, + FastModeCommand::Off => false, + FastModeCommand::Status => unreachable!(), + }; + crate::config::save_fast_mode_preference(&config.root, enabled)?; + config.default_fast_mode = enabled; + Ok(fast_mode_change_message(&config.fast_mode_state())) + } + } +} + +fn fast_mode_change_message(state: &FastModeState) -> String { + if state.active { + "fast mode enabled".into() + } else if state.preferred { + format!( + "fast mode preference on; unavailable for {}", + state + .unavailable_reason + .as_deref() + .unwrap_or("this provider/model") + ) + } else { + "fast mode off".into() + } +} + +fn fast_mode_status(state: &FastModeState) -> String { + if state.active { + "enabled".into() + } else if state.preferred { + format!( + "preferred, unavailable for {}", + state + .unavailable_reason + .as_deref() + .unwrap_or("this provider/model") + ) + } else { + "off".into() + } +} + +fn model_status_message(config: &Config) -> String { + let model = &config.model; + let state = config.fast_mode_state(); + if state.active { + format!("model: {model} · fast enabled") + } else if state.preferred { + format!("model: {model} · fast unavailable") + } else { + format!("model: {model}") + } +} + fn transcript_from_loaded( conversation: &Conversation, warning: Option, @@ -2667,6 +2818,67 @@ mod tests { ); } + #[test] + fn parse_local_command_accepts_fast_forms() { + assert_eq!( + parse_local_command("/fast").unwrap(), + LocalCommand::Fast(FastModeCommand::Toggle) + ); + assert_eq!( + parse_local_command("/fast on").unwrap(), + LocalCommand::Fast(FastModeCommand::On) + ); + assert_eq!( + parse_local_command("/fast off").unwrap(), + LocalCommand::Fast(FastModeCommand::Off) + ); + assert_eq!( + parse_local_command("/fast status").unwrap(), + LocalCommand::Fast(FastModeCommand::Status) + ); + assert_eq!( + parse_local_command("/fast maybe"), + Err("usage: /fast [on|off|status]".into()) + ); + } + + #[test] + fn fast_mode_status_distinguishes_preference_and_activation() { + let mut config = Config { + default_fast_mode: true, + ..Config::default() + }; + + assert_eq!( + fast_mode_status(&config.fast_mode_state()), + "preferred, unavailable for provider fireworks" + ); + + config.provider_id = config::CHATGPT_CODEX_PROVIDER_ID.into(); + config.active_provider.kind = config::CHATGPT_CODEX_PROVIDER_KIND.into(); + config.model = config::CHATGPT_CODEX_DEFAULT_MODEL.into(); + config.model_metadata = Some(config::ModelDefinition { + id: config::CHATGPT_CODEX_DEFAULT_MODEL.into(), + provider: config::CHATGPT_CODEX_PROVIDER_ID.into(), + display_name: None, + context_length: None, + max_output_tokens: None, + supports_tools: true, + supports_streaming: true, + reasoning: Default::default(), + fast_mode: config::FastModeMetadata { supported: true }, + }); + + assert_eq!(fast_mode_status(&config.fast_mode_state()), "enabled"); + assert_eq!( + model_status_message(&config), + format!( + "model: {} · fast enabled", + config::CHATGPT_CODEX_DEFAULT_MODEL + ) + ); + } + #[test] fn cancelled_turn_repairs_missing_tool_results() { let root = tempdir().unwrap(); diff --git a/src/config.rs b/src/config.rs index a92fdef..863d3de 100644 --- a/src/config.rs +++ b/src/config.rs @@ -27,6 +27,8 @@ pub struct ConfigFile { pub default_model: Option, #[serde(skip_serializing_if = "Option::is_none")] pub default_reasoning_effort: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub default_fast_mode: Option, // Deprecated compatibility fields accepted from older config.json files. #[serde(skip_serializing_if = "Option::is_none")] @@ -97,6 +99,8 @@ pub struct ModelDefinition { pub supports_streaming: bool, #[serde(default)] pub reasoning: ReasoningMetadata, + #[serde(default)] + pub fast_mode: FastModeMetadata, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -112,6 +116,13 @@ pub struct ReasoningMetadata { pub request_format: ReasoningRequestFormat, } +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct FastModeMetadata { + #[serde(default)] + pub supported: bool, +} + #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "lowercase")] pub enum ReasoningEffort { @@ -145,6 +156,7 @@ pub struct Config { pub provider_id: String, pub model: String, pub reasoning_effort: ReasoningEffort, + pub default_fast_mode: bool, pub active_provider: ResolvedProviderConfig, pub model_metadata: Option, pub default_access_mode: AccessMode, @@ -157,6 +169,14 @@ pub struct Config { pub docs_dir: PathBuf, } +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct FastModeState { + pub preferred: bool, + pub supported: bool, + pub active: bool, + pub unavailable_reason: Option, +} + #[derive(Debug, Clone, Default)] pub struct ConfigOverrides { pub model: Option, @@ -202,6 +222,12 @@ impl Default for ReasoningMetadata { } } +impl Default for FastModeMetadata { + fn default() -> Self { + Self { supported: false } + } +} + impl Default for ReasoningRequestFormat { fn default() -> Self { Self::ReasoningEffort @@ -287,6 +313,7 @@ impl Default for Config { provider_id: DEFAULT_PROVIDER_ID.to_string(), model: DEFAULT_MODEL.to_string(), reasoning_effort: ReasoningEffort::Medium, + default_fast_mode: false, active_provider, model_metadata: Some(default_model_definition()), default_access_mode: AccessMode::ReadOnly, @@ -374,6 +401,9 @@ impl Config { if let Some(v) = file.confirm_destructive_operations { cfg.confirm_destructive_operations = v; } + if let Some(v) = file.default_fast_mode { + cfg.default_fast_mode = v; + } } if let Some(access_mode) = overrides.access_mode { @@ -443,6 +473,43 @@ impl Config { kind => bail!("unsupported provider kind `{kind}`"), } } + + pub fn fast_mode_state(&self) -> FastModeState { + let preferred = self.default_fast_mode; + let supported = self.fast_mode_supported(); + let active = preferred && supported; + let unavailable_reason = if preferred && !supported { + Some(self.fast_mode_unavailable_reason()) + } else { + None + }; + FastModeState { + preferred, + supported, + active, + unavailable_reason, + } + } + + fn fast_mode_supported(&self) -> bool { + if self + .model_metadata + .as_ref() + .is_some_and(|model| model.fast_mode.supported) + { + return true; + } + + self.model_metadata.is_none() && self.active_provider.kind == CHATGPT_CODEX_PROVIDER_KIND + } + + fn fast_mode_unavailable_reason(&self) -> String { + if self.active_provider.kind != CHATGPT_CODEX_PROVIDER_KIND { + format!("provider {}", self.provider_id) + } else { + format!("model {}", self.model) + } + } } impl ProviderDefinition { @@ -483,6 +550,28 @@ pub fn save_last_used(root: &Path, model: &str, reasoning_effort: ReasoningEffor file.default_reasoning_effort = Some(reasoning_effort); write_json_pretty(&path, &file) } + +pub fn save_last_used_provider( + root: &Path, + provider_id: &str, + model: &str, + reasoning_effort: ReasoningEffort, +) -> Result<()> { + let path = config_path(root); + let mut file = load_config_file(root)?.unwrap_or_default(); + file.default_provider = Some(provider_id.to_string()); + file.default_model = Some(model.to_string()); + file.default_reasoning_effort = Some(reasoning_effort); + write_json_pretty(&path, &file) +} + +pub fn save_fast_mode_preference(root: &Path, enabled: bool) -> Result<()> { + let path = config_path(root); + let mut file = load_config_file(root)?.unwrap_or_default(); + file.default_fast_mode = Some(enabled); + write_json_pretty(&path, &file) +} + pub fn load_or_create_default_provider_registry(root: &Path) -> Result { fs::create_dir_all(root).with_context(|| format!("creating {}", root.display()))?; let path = providers_path(root); @@ -533,6 +622,7 @@ pub fn default_model_definition() -> ModelDefinition { supports_tools: true, supports_streaming: true, reasoning: ReasoningMetadata::default(), + fast_mode: FastModeMetadata::default(), } } diff --git a/src/providers/chatgpt_codex.rs b/src/providers/chatgpt_codex.rs index 255e17e..cf6c6bd 100644 --- a/src/providers/chatgpt_codex.rs +++ b/src/providers/chatgpt_codex.rs @@ -17,6 +17,7 @@ pub struct ChatGptCodexProvider { model: String, endpoint: String, reasoning_effort: ReasoningEffort, + fast_mode: bool, } #[derive(Debug, Clone)] @@ -24,6 +25,7 @@ pub struct ChatGptCodexSettings { pub model: String, pub endpoint: String, pub reasoning_effort: ReasoningEffort, + pub fast_mode: bool, } #[derive(Debug, Default, Clone)] @@ -40,6 +42,7 @@ impl ChatGptCodexProvider { model: settings.model, endpoint: normalize_endpoint(&settings.endpoint), reasoning_effort: settings.reasoning_effort, + fast_mode: settings.fast_mode, } } @@ -51,7 +54,13 @@ impl ChatGptCodexProvider { ) -> Result { let token = load_codex_access_token()?; let secret = token.as_secret().to_string(); - let body = responses_body(&self.model, messages, tools, self.reasoning_effort); + let body = responses_body( + &self.model, + messages, + tools, + self.reasoning_effort, + self.fast_mode, + ); let resp = self .client .post(&self.endpoint) @@ -128,6 +137,7 @@ fn responses_body( messages: Vec, tools: Vec, reasoning_effort: ReasoningEffort, + fast_mode: bool, ) -> Value { let mut instructions = Vec::new(); let mut input = Vec::new(); @@ -183,7 +193,9 @@ fn responses_body( if !instructions.is_empty() { body["instructions"] = Value::String(instructions.join("\n\n")); } - if let Some(effort) = reasoning_effort.request_value() { + if fast_mode { + body["reasoning"] = json!({"effort": "minimal", "summary": "auto"}); + } else if let Some(effort) = reasoning_effort.request_value() { body["reasoning"] = json!({"effort": effort, "summary": "auto"}); } body @@ -464,6 +476,7 @@ mod tests { ], Vec::new(), ReasoningEffort::Off, + false, ); assert_eq!(body["model"], "gpt-test"); @@ -473,6 +486,24 @@ mod tests { })); } + #[test] + fn responses_body_uses_minimal_reasoning_for_fast_mode() { + let body = responses_body( + "gpt-test", + vec![ModelMessage::User { + content: "hello".into(), + }], + Vec::new(), + ReasoningEffort::High, + true, + ); + + assert_eq!( + body["reasoning"], + json!({"effort": "minimal", "summary": "auto"}) + ); + } + #[test] fn stream_parser_collects_text_and_function_call() { let (tx, _rx) = mpsc::unbounded_channel(); diff --git a/src/providers/mod.rs b/src/providers/mod.rs index eecaa4c..1f67008 100644 --- a/src/providers/mod.rs +++ b/src/providers/mod.rs @@ -17,8 +17,14 @@ pub enum ProviderClient { ChatGptCodex(ChatGptCodexProvider), } +#[derive(Debug, Clone, Copy)] +pub struct ProviderRuntimeOptions { + pub reasoning_effort: ReasoningEffort, + pub fast_mode: bool, +} + impl ProviderClient { - pub fn from_config(config: &Config, reasoning_effort: ReasoningEffort) -> Result { + pub fn from_config(config: &Config, options: ProviderRuntimeOptions) -> Result { match config.active_provider.kind.as_str() { DEFAULT_PROVIDER_KIND => { let api_key = config.resolved_api_key()?; @@ -32,7 +38,7 @@ impl ProviderClient { model: config.model.clone(), base_url: config.active_provider.base_url.clone(), api_key, - reasoning_effort, + reasoning_effort: options.reasoning_effort, reasoning_request_format, }, ))) @@ -41,7 +47,8 @@ impl ProviderClient { ChatGptCodexSettings { model: config.model.clone(), endpoint: config.active_provider.base_url.clone(), - reasoning_effort, + reasoning_effort: options.reasoning_effort, + fast_mode: options.fast_mode, }, ))), kind => bail!("unsupported provider kind `{kind}`"), diff --git a/src/setup.rs b/src/setup.rs index e59d2bd..1b05586 100644 --- a/src/setup.rs +++ b/src/setup.rs @@ -2,10 +2,10 @@ use crate::check; use crate::cli::Cli; use crate::codex_auth; use crate::config::{ - self, ConfigFile, ModelDefinition, ModelsFile, ProviderDefinition, ProvidersFile, - ReasoningEffort, ReasoningMetadata, ReasoningRequestFormat, CHATGPT_CODEX_DEFAULT_MODEL, - CHATGPT_CODEX_PROVIDER_ID, CHATGPT_CODEX_PROVIDER_KIND, CHATGPT_CODEX_PROVIDER_NAME, - CHATGPT_CODEX_RESPONSES_URL, DEFAULT_PROVIDER_KIND, + self, ConfigFile, FastModeMetadata, ModelDefinition, ModelsFile, ProviderDefinition, + ProvidersFile, ReasoningEffort, ReasoningMetadata, ReasoningRequestFormat, + CHATGPT_CODEX_DEFAULT_MODEL, CHATGPT_CODEX_PROVIDER_ID, CHATGPT_CODEX_PROVIDER_KIND, + CHATGPT_CODEX_PROVIDER_NAME, CHATGPT_CODEX_RESPONSES_URL, DEFAULT_PROVIDER_KIND, }; use crate::menu::{Menu, MenuItem, TextPrompt}; use anyhow::{bail, Context, Result}; @@ -970,6 +970,9 @@ fn upsert_model(models: &mut ModelsFile, selection: &SetupSelection) { }, request_format: ReasoningRequestFormat::ReasoningEffort, }, + fast_mode: FastModeMetadata { + supported: is_chatgpt_codex_provider(&selection.provider_id), + }, }; if let Some(existing) = models.models.iter_mut().find(|existing| { diff --git a/src/ui/render.rs b/src/ui/render.rs index 5a351e8..181c24c 100644 --- a/src/ui/render.rs +++ b/src/ui/render.rs @@ -53,6 +53,7 @@ pub struct RenderState<'a> { pub show_full_tools: bool, pub show_reasoning: bool, pub reasoning_effort: ReasoningEffort, + pub fast_mode_active: bool, pub scroll: u16, pub autofill: Option<&'a AutoFillMenu>, pub overlay: Option<&'a OverlayView>, @@ -724,6 +725,9 @@ fn footer_text(state: &RenderState<'_>) -> String { parts.push("tools:full".into()); } parts.push(format!("reasoning:{}", state.reasoning_effort)); + if state.fast_mode_active { + parts.push("fast".into()); + } if state.show_reasoning { parts.push("reasoning:visible".into()); } diff --git a/tests/agent_tests.rs b/tests/agent_tests.rs index 3964842..b5059b6 100644 --- a/tests/agent_tests.rs +++ b/tests/agent_tests.rs @@ -123,6 +123,62 @@ async fn reasoning_effort_supports_reasoning_object_format() { .unwrap(); } +#[tokio::test] +async fn fast_mode_preference_does_not_change_openai_compatible_request() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/chat/completions")) + .respond_with(sse( + "data: {\"choices\":[{\"index\":0,\"delta\":{\"content\":\"Done.\"}}]}\r\n\r\ndata: [DONE]\r\n\r\n", + )) + .expect(1) + .mount(&server) + .await; + + let root = tempdir().unwrap(); + let cwd = tempdir().unwrap(); + let docs = tempdir().unwrap(); + let config = Config { + root: root.path().to_path_buf(), + docs_dir: docs.path().to_path_buf(), + model: "test-model".into(), + default_fast_mode: true, + active_provider: cassady::config::ResolvedProviderConfig { + base_url: server.uri(), + api_key: "test-key".into(), + ..Config::default().active_provider + }, + ..Config::default() + }; + let conversation = Conversation::create( + &config.conversations_dir(), + &config.model, + cwd.path(), + "base prompt".into(), + ) + .unwrap(); + let (tx, _rx) = mpsc::unbounded_channel::(); + + run_turn( + conversation, + "stay compatible".into(), + AgentSettings { + config, + cwd: cwd.path().to_path_buf(), + mode: AccessMode::ReadOnly, + reasoning_effort: ReasoningEffort::Off, + }, + tx, + ) + .await + .unwrap(); + + let requests = server.received_requests().await.unwrap(); + let body = String::from_utf8_lossy(&requests[0].body); + assert!(!body.contains("\"effort\":\"minimal\"")); + assert!(!body.contains("\"fast_mode\"")); +} + #[tokio::test] async fn reasoning_is_streamed_persisted_and_sent_back() { let server = MockServer::start().await; diff --git a/tests/config_tests.rs b/tests/config_tests.rs index 37589e3..c908086 100644 --- a/tests/config_tests.rs +++ b/tests/config_tests.rs @@ -47,6 +47,7 @@ fn default_provider_and_model_files_are_created() { .unwrap(); assert_eq!(models.models[0].provider, "fireworks"); assert!(models.models[0].reasoning.supported); + assert!(!models.models[0].fast_mode.supported); assert_eq!( models.models[0].reasoning.default_effort, ReasoningEffort::Medium @@ -127,6 +128,116 @@ fn reasoning_defaults_to_supported_medium_for_model_metadata() { ); } +#[test] +fn fast_mode_defaults_to_off_and_unsupported() { + let model: config::ModelDefinition = serde_json::from_str( + r#"{ + "id": "test-model", + "provider": "test-provider" +} +"#, + ) + .unwrap(); + + assert!(!model.fast_mode.supported); + + let root = tempdir().unwrap(); + let cfg = Config::load_from_root_with_docs( + root.path().to_path_buf(), + root.path().join("docs"), + &cli(), + ) + .unwrap(); + let state = cfg.fast_mode_state(); + assert!(!state.preferred); + assert!(!state.supported); + assert!(!state.active); +} + +#[test] +fn fast_mode_preference_persists_without_losing_config_fields() { + let root = tempdir().unwrap(); + std::fs::write( + root.path().join("config.json"), + r#"{ + "default_access_mode": "workspace-edit", + "show_reasoning": true +} +"#, + ) + .unwrap(); + + config::save_fast_mode_preference(root.path(), true).unwrap(); + + let cfg = Config::load_from_root_with_docs( + root.path().to_path_buf(), + root.path().join("docs"), + &cli(), + ) + .unwrap(); + assert!(cfg.default_fast_mode); + assert!(cfg.show_reasoning); + assert_eq!(cfg.default_access_mode.to_string(), "workspace-edit"); +} + +#[test] +fn fast_mode_state_is_active_for_supported_codex_model() { + let root = tempdir().unwrap(); + std::fs::write( + root.path().join("providers.json"), + r#"{ + "providers": [ + { + "id": "chatgpt-codex", + "name": "ChatGPT Codex", + "kind": "chatgpt-codex", + "base_url": "https://chatgpt.com/backend-api/codex/responses", + "api_key": "", + "default_model": "gpt-5.5", + "models": ["gpt-5.5"] + } + ] +} +"#, + ) + .unwrap(); + std::fs::write( + root.path().join("models.json"), + r#"{ + "models": [ + { + "id": "gpt-5.5", + "provider": "chatgpt-codex", + "fast_mode": { "supported": true } + } + ] +} +"#, + ) + .unwrap(); + std::fs::write( + root.path().join("config.json"), + r#"{ + "default_provider": "chatgpt-codex", + "default_model": "gpt-5.5", + "default_fast_mode": true +} +"#, + ) + .unwrap(); + + let cfg = Config::load_from_root_with_docs( + root.path().to_path_buf(), + root.path().join("docs"), + &cli(), + ) + .unwrap(); + let state = cfg.fast_mode_state(); + assert!(state.preferred); + assert!(state.supported); + assert!(state.active); +} + #[test] fn validation_accepts_chatgpt_codex_without_api_key() { let providers = ProvidersFile { @@ -150,6 +261,7 @@ fn validation_accepts_chatgpt_codex_without_api_key() { supports_tools: true, supports_streaming: true, reasoning: Default::default(), + fast_mode: Default::default(), }], }; diff --git a/tests/setup_tests.rs b/tests/setup_tests.rs index 5a90811..bf320f8 100644 --- a/tests/setup_tests.rs +++ b/tests/setup_tests.rs @@ -174,6 +174,11 @@ fn apply_setup_writes_chatgpt_codex_without_api_key() { assert_eq!(provider.id, config::CHATGPT_CODEX_PROVIDER_ID); assert_eq!(provider.kind, config::CHATGPT_CODEX_PROVIDER_KIND); assert!(provider.api_key.is_empty()); + + let models: ModelsFile = + serde_json::from_str(&std::fs::read_to_string(root.path().join("models.json")).unwrap()) + .unwrap(); + assert!(models.models[0].fast_mode.supported); } #[test]