Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b35c4fb7e1 | ||
|
|
d05bb87a23 |
Generated
+1
-1
@@ -226,7 +226,7 @@ checksum = "8ae3f5d315924270530207e2a68396c3cc547f6dca3fbdca317cfb1a51edb593"
|
||||
|
||||
[[package]]
|
||||
name = "cassady"
|
||||
version = "0.3.2"
|
||||
version = "0.3.4"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-trait",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "cassady"
|
||||
version = "0.3.2"
|
||||
version = "0.3.4"
|
||||
edition = "2021"
|
||||
description = "Cassady/Cass minimal terminal coding agent"
|
||||
license = "MIT"
|
||||
|
||||
@@ -80,7 +80,7 @@ Common in-chat commands:
|
||||
- `/branch` or `/restore`: open the branch/restore menu.
|
||||
- `/login`: configure or update provider login settings.
|
||||
- `/logout`: remove saved provider config and associated model entries.
|
||||
- `/fast`, `/fast on`, `/fast off`, `/fast status`: prefer faster inference when the active provider/model supports it. v0.3.2 supports this for ChatGPT Codex models marked fast-capable.
|
||||
- `/fast`, `/fast on`, `/fast off`, `/fast status`: prefer faster inference when the active provider/model supports it. ChatGPT Codex models, including `gpt-5.5`, are treated as fast-capable.
|
||||
- `/model <model>`: switch to a model from `~/.cass/models.json`.
|
||||
- `/new`: create a new chat for the current directory.
|
||||
- `/resume <chat>`: resume a saved chat for the current directory.
|
||||
|
||||
+18
-2
@@ -1,8 +1,24 @@
|
||||
# Cassady (Cass) Roadmap
|
||||
|
||||
## v0.3.3 — Tool Output Context Reliability
|
||||
## v0.3.3 — Codex Fast-Mode Compatibility
|
||||
|
||||
This release focuses on making large tool outputs easier for the assistant to recover from when model-context compaction or truncation hides important details. Cassady should guide the assistant toward smaller, targeted reads and searches, preserve enough provenance for follow-up inspection, and add regression coverage for broad-output workflows that previously stalled safe edits. See `plans/V0_3_3_TOOL_OUTPUT_CONTEXT_RELIABILITY_PLAN.md`.
|
||||
This release focuses on keeping fast mode available for ChatGPT Codex users when local model metadata predates the fast-mode capability flag. Cassady should treat active `chatgpt-codex` provider models, including `gpt-5.5`, as fast-capable while leaving OpenAI-compatible and custom providers capability-gated by model metadata.
|
||||
|
||||
### Fast Mode Compatibility
|
||||
|
||||
- [x] **Treat ChatGPT Codex models as fast-capable at runtime.** Make `/fast` active for any active `chatgpt-codex` provider model even when legacy `models.json` metadata says `fast_mode.supported` is false.
|
||||
- Keep non-Codex providers governed by their model metadata.
|
||||
- Preserve the saved fast-mode preference behavior and status reporting.
|
||||
|
||||
- [x] **Document the Codex capability fallback.** Update README and bundled docs so users understand that ChatGPT Codex models, including `gpt-5.5`, can honor fast mode without refreshed metadata.
|
||||
- Keep docs clear that provider-specific fast-mode request shaping remains Codex-only.
|
||||
|
||||
- [x] **Add regression coverage for legacy metadata.** Test that a ChatGPT Codex model with older `fast_mode.supported: false` metadata still reports fast mode as supported and active when preferred.
|
||||
- Verify `cargo fmt` and `cargo test --locked --all-targets` pass before handoff.
|
||||
|
||||
## v0.3.4 — Tool Output Context Reliability
|
||||
|
||||
This release focuses on making large tool outputs easier for the assistant to recover from when model-context compaction or truncation hides important details. Cassady should guide the assistant toward smaller, targeted reads and searches, preserve enough provenance for follow-up inspection, and add regression coverage for broad-output workflows that previously stalled safe edits. See `plans/V0_3_4_TOOL_OUTPUT_CONTEXT_RELIABILITY_PLAN.md`.
|
||||
|
||||
### Model Context Recovery
|
||||
|
||||
|
||||
+1
-1
@@ -107,7 +107,7 @@ The updater does not invoke `sudo` or administrator prompts. If the install dire
|
||||
Type `/` to open command autocomplete.
|
||||
|
||||
- `/branch` or `/restore`: open the branch/restore menu for the current conversation family.
|
||||
- `/fast`, `/fast on`, `/fast off`, `/fast status`: toggle or inspect a persisted fast-mode preference. Fast mode is active only when the current provider/model supports it; v0.3.2 supports ChatGPT Codex models marked fast-capable.
|
||||
- `/fast`, `/fast on`, `/fast off`, `/fast status`: toggle or inspect a persisted fast-mode preference. Fast mode is active only when the current provider/model supports it; ChatGPT Codex models, including `gpt-5.5`, are treated as fast-capable.
|
||||
- `/login`: configure or update provider login settings, then reload active provider/model config.
|
||||
- `/logout`: remove saved providers and their associated models, then reload active provider/model config when any remain.
|
||||
- `/model <model>`: switch the model for future turns. Autocomplete lists models from `~/.cass/models.json`.
|
||||
|
||||
@@ -162,11 +162,11 @@ Fields:
|
||||
- `default_effort`: optional `off`, `low`, `medium`, or `high`; defaults to `medium`. Cannot effectively be `off` when `required` is `true`.
|
||||
- `request_format`: optional `reasoning_effort` or `reasoning_object`; defaults to `reasoning_effort`.
|
||||
- `fast_mode`: optional object. Defaults to unsupported.
|
||||
- `supported`: optional boolean, defaults to `false`. Setup marks ChatGPT Codex model entries as supported; custom and OpenAI-compatible model entries default to unsupported.
|
||||
- `supported`: optional boolean, defaults to `false`. Cassady treats active `chatgpt-codex` provider models as fast-capable even if older metadata says otherwise; custom and OpenAI-compatible model entries default to unsupported.
|
||||
|
||||
Reasoning effort is a runtime per-turn setting. Press `Tab` to cycle it while idle. Provider-streamed reasoning is persisted and sent back in future model context using the provider's reasoning field, such as `reasoning_content` or `reasoning`.
|
||||
|
||||
Fast mode is a persisted preference, not a guarantee. Use `/fast` to toggle it while idle. `/status` shows `enabled` only when the preference is on and the current provider/model can honor it; otherwise it reports `off` or `preferred, unavailable ...`. In v0.3.2, fast-mode request shaping is implemented only for `chatgpt-codex`.
|
||||
Fast mode is a persisted preference, not a guarantee. Use `/fast` to toggle it while idle. `/status` shows `enabled` only when the preference is on and the current provider/model can honor it; otherwise it reports `off` or `preferred, unavailable ...`. Fast-mode request shaping is implemented only for `chatgpt-codex`.
|
||||
|
||||
## Precedence
|
||||
|
||||
|
||||
+2
-2
@@ -96,9 +96,9 @@ Reasoning display is separate. `show_reasoning` controls whether provider-stream
|
||||
Fast mode has two parts:
|
||||
|
||||
- `default_fast_mode` in `config.json`: the user's saved preference.
|
||||
- `fast_mode.supported` in `models.json`: whether the active provider/model can honor that preference.
|
||||
- `fast_mode.supported` in `models.json`: whether non-Codex provider/model metadata can honor that preference.
|
||||
|
||||
In v0.3.2, Cassady sends fast-mode requests only for `ChatGPT Codex`. Setup marks ChatGPT Codex model entries as fast-capable. OpenAI-compatible and custom model entries default to unsupported, so `/fast` can remember the preference without sending provider-specific fields.
|
||||
Cassady sends fast-mode requests only for `ChatGPT Codex`. Any active `chatgpt-codex` provider model, including `gpt-5.5`, is treated as fast-capable so older model metadata does not block the feature. OpenAI-compatible and custom model entries default to unsupported, so `/fast` can remember the preference without sending provider-specific fields.
|
||||
|
||||
## Switching models
|
||||
|
||||
|
||||
+2
-2
@@ -111,9 +111,9 @@ Inside a chat:
|
||||
/fast
|
||||
```
|
||||
|
||||
Use `/fast on`, `/fast off`, or `/fast status` when you want an explicit action. Cassady saves the preference in `config.json`, but fast mode is active only when the current provider/model supports it. In v0.3.2, support is implemented for ChatGPT Codex model entries marked with `fast_mode.supported`.
|
||||
Use `/fast on`, `/fast off`, or `/fast status` when you want an explicit action. Cassady saves the preference in `config.json`, but fast mode is active only when the current provider/model supports it. ChatGPT Codex models, including `gpt-5.5`, are treated as fast-capable.
|
||||
|
||||
Switching to an unsupported provider/model keeps the preference but makes `/status` show fast mode as unavailable. Switching back to a supported ChatGPT Codex model enables it again.
|
||||
Switching to an unsupported provider/model keeps the preference but makes `/status` show fast mode as unavailable. Switching back to ChatGPT Codex enables it again.
|
||||
|
||||
## Resume a chat
|
||||
|
||||
|
||||
+2
-2
@@ -1,8 +1,8 @@
|
||||
# v0.3.3 Tool Output Context Reliability Implementation Plan
|
||||
# v0.3.4 Tool Output Context Reliability Implementation Plan
|
||||
|
||||
## Goal
|
||||
|
||||
v0.3.3 makes Cassady more reliable after broad tool output has been truncated, compacted, or superseded in the model context. The assistant should be able to tell when details are missing, understand which file range or command produced them, and quickly recover by using narrower reads or searches instead of stalling or making unsafe edits from incomplete context.
|
||||
v0.3.4 makes Cassady more reliable after broad tool output has been truncated, compacted, or superseded in the model context. The assistant should be able to tell when details are missing, understand which file range or command produced them, and quickly recover by using narrower reads or searches instead of stalling or making unsafe edits from incomplete context.
|
||||
|
||||
Success statement:
|
||||
|
||||
+4
-6
@@ -492,15 +492,13 @@ impl Config {
|
||||
}
|
||||
|
||||
fn fast_mode_supported(&self) -> bool {
|
||||
if self
|
||||
.model_metadata
|
||||
.as_ref()
|
||||
.is_some_and(|model| model.fast_mode.supported)
|
||||
{
|
||||
if self.active_provider.kind == CHATGPT_CODEX_PROVIDER_KIND {
|
||||
return true;
|
||||
}
|
||||
|
||||
self.model_metadata.is_none() && self.active_provider.kind == CHATGPT_CODEX_PROVIDER_KIND
|
||||
self.model_metadata
|
||||
.as_ref()
|
||||
.is_some_and(|model| model.fast_mode.supported)
|
||||
}
|
||||
|
||||
fn fast_mode_unavailable_reason(&self) -> String {
|
||||
|
||||
@@ -197,6 +197,8 @@ fn responses_body(
|
||||
body["reasoning"] = json!({"effort": "minimal", "summary": "auto"});
|
||||
} else if let Some(effort) = reasoning_effort.request_value() {
|
||||
body["reasoning"] = json!({"effort": effort, "summary": "auto"});
|
||||
} else if reasoning_effort == ReasoningEffort::Off {
|
||||
body["reasoning"] = json!({"effort": "none", "summary": "auto"});
|
||||
}
|
||||
body
|
||||
}
|
||||
@@ -504,6 +506,24 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn responses_body_sends_none_effort_when_reasoning_is_off() {
|
||||
let body = responses_body(
|
||||
"gpt-test",
|
||||
vec![ModelMessage::User {
|
||||
content: "hello".into(),
|
||||
}],
|
||||
Vec::new(),
|
||||
ReasoningEffort::Off,
|
||||
false,
|
||||
);
|
||||
|
||||
assert_eq!(
|
||||
body["reasoning"],
|
||||
json!({"effort": "none", "summary": "auto"})
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stream_parser_collects_text_and_function_call() {
|
||||
let (tx, _rx) = mpsc::unbounded_channel();
|
||||
|
||||
@@ -28,11 +28,13 @@ impl ProviderClient {
|
||||
match config.active_provider.kind.as_str() {
|
||||
DEFAULT_PROVIDER_KIND => {
|
||||
let api_key = config.resolved_api_key()?;
|
||||
let reasoning_request_format = config
|
||||
.model_metadata
|
||||
.as_ref()
|
||||
let model_metadata = config.model_metadata.as_ref();
|
||||
let reasoning_request_format = model_metadata
|
||||
.map(|model| model.reasoning.request_format)
|
||||
.unwrap_or_default();
|
||||
let reasoning_supported = model_metadata
|
||||
.map(|model| model.reasoning.supported)
|
||||
.unwrap_or(false);
|
||||
Ok(Self::OpenAiCompatible(OpenAiCompatibleProvider::new(
|
||||
OpenAiCompatibleSettings {
|
||||
model: config.model.clone(),
|
||||
@@ -40,6 +42,7 @@ impl ProviderClient {
|
||||
api_key,
|
||||
reasoning_effort: options.reasoning_effort,
|
||||
reasoning_request_format,
|
||||
reasoning_supported,
|
||||
},
|
||||
)))
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ pub struct OpenAiCompatibleProvider {
|
||||
api_key: String,
|
||||
reasoning_effort: ReasoningEffort,
|
||||
reasoning_request_format: ReasoningRequestFormat,
|
||||
reasoning_supported: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
@@ -27,6 +28,7 @@ pub struct OpenAiCompatibleSettings {
|
||||
pub api_key: String,
|
||||
pub reasoning_effort: ReasoningEffort,
|
||||
pub reasoning_request_format: ReasoningRequestFormat,
|
||||
pub reasoning_supported: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
@@ -45,6 +47,7 @@ impl OpenAiCompatibleProvider {
|
||||
api_key: settings.api_key,
|
||||
reasoning_effort: settings.reasoning_effort,
|
||||
reasoning_request_format: settings.reasoning_request_format,
|
||||
reasoning_supported: settings.reasoning_supported,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -65,6 +68,7 @@ impl OpenAiCompatibleProvider {
|
||||
&mut body,
|
||||
self.reasoning_effort,
|
||||
self.reasoning_request_format,
|
||||
self.reasoning_supported,
|
||||
);
|
||||
let resp = self
|
||||
.client
|
||||
@@ -219,9 +223,17 @@ fn apply_reasoning_request(
|
||||
body: &mut Value,
|
||||
effort: ReasoningEffort,
|
||||
format: ReasoningRequestFormat,
|
||||
supported: bool,
|
||||
) {
|
||||
let Some(effort) = effort.request_value() else {
|
||||
if !supported {
|
||||
return;
|
||||
}
|
||||
let effort_str = match effort {
|
||||
ReasoningEffort::Off => "none",
|
||||
_ => match effort.request_value() {
|
||||
Some(value) => value,
|
||||
None => return,
|
||||
},
|
||||
};
|
||||
let Value::Object(obj) = body else {
|
||||
return;
|
||||
@@ -230,11 +242,11 @@ fn apply_reasoning_request(
|
||||
ReasoningRequestFormat::ReasoningEffort => {
|
||||
obj.insert(
|
||||
"reasoning_effort".to_string(),
|
||||
Value::String(effort.to_string()),
|
||||
Value::String(effort_str.to_string()),
|
||||
);
|
||||
}
|
||||
ReasoningRequestFormat::ReasoningObject => {
|
||||
obj.insert("reasoning".to_string(), json!({ "effort": effort }));
|
||||
obj.insert("reasoning".to_string(), json!({ "effort": effort_str }));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -316,3 +328,60 @@ fn chat_url(base: &str) -> String {
|
||||
format!("{}/chat/completions", base.trim_end_matches('/'))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn reasoning_effort_format_sends_none_when_off_and_supported() {
|
||||
let mut body = json!({"model": "test"});
|
||||
apply_reasoning_request(
|
||||
&mut body,
|
||||
ReasoningEffort::Off,
|
||||
ReasoningRequestFormat::ReasoningEffort,
|
||||
true,
|
||||
);
|
||||
assert_eq!(
|
||||
body["reasoning_effort"],
|
||||
Value::String("none".to_string())
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_object_format_sends_none_when_off_and_supported() {
|
||||
let mut body = json!({"model": "test"});
|
||||
apply_reasoning_request(
|
||||
&mut body,
|
||||
ReasoningEffort::Off,
|
||||
ReasoningRequestFormat::ReasoningObject,
|
||||
true,
|
||||
);
|
||||
assert_eq!(body["reasoning"], json!({ "effort": "none" }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_sends_nothing_when_unsupported_even_if_off() {
|
||||
let mut body = json!({"model": "test"});
|
||||
apply_reasoning_request(
|
||||
&mut body,
|
||||
ReasoningEffort::Off,
|
||||
ReasoningRequestFormat::ReasoningEffort,
|
||||
false,
|
||||
);
|
||||
assert!(body.get("reasoning_effort").is_none());
|
||||
assert!(body.get("reasoning").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_sends_nothing_when_unsupported_even_if_high() {
|
||||
let mut body = json!({"model": "test"});
|
||||
apply_reasoning_request(
|
||||
&mut body,
|
||||
ReasoningEffort::High,
|
||||
ReasoningRequestFormat::ReasoningObject,
|
||||
false,
|
||||
);
|
||||
assert!(body.get("reasoning").is_none());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -238,6 +238,64 @@ fn fast_mode_state_is_active_for_supported_codex_model() {
|
||||
assert!(state.active);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fast_mode_state_is_active_for_chatgpt_codex_even_with_legacy_metadata() {
|
||||
let root = tempdir().unwrap();
|
||||
std::fs::write(
|
||||
root.path().join("providers.json"),
|
||||
r#"{
|
||||
"providers": [
|
||||
{
|
||||
"id": "chatgpt-codex",
|
||||
"name": "ChatGPT Codex",
|
||||
"kind": "chatgpt-codex",
|
||||
"base_url": "https://chatgpt.com/backend-api/codex/responses",
|
||||
"api_key": "",
|
||||
"default_model": "gpt-5.5",
|
||||
"models": ["gpt-5.5"]
|
||||
}
|
||||
]
|
||||
}
|
||||
"#,
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
root.path().join("models.json"),
|
||||
r#"{
|
||||
"models": [
|
||||
{
|
||||
"id": "gpt-5.5",
|
||||
"provider": "chatgpt-codex",
|
||||
"fast_mode": { "supported": false }
|
||||
}
|
||||
]
|
||||
}
|
||||
"#,
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
root.path().join("config.json"),
|
||||
r#"{
|
||||
"default_provider": "chatgpt-codex",
|
||||
"default_model": "gpt-5.5",
|
||||
"default_fast_mode": true
|
||||
}
|
||||
"#,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let cfg = Config::load_from_root_with_docs(
|
||||
root.path().to_path_buf(),
|
||||
root.path().join("docs"),
|
||||
&cli(),
|
||||
)
|
||||
.unwrap();
|
||||
let state = cfg.fast_mode_state();
|
||||
assert!(state.preferred);
|
||||
assert!(state.supported);
|
||||
assert!(state.active);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validation_accepts_chatgpt_codex_without_api_key() {
|
||||
let providers = ProvidersFile {
|
||||
|
||||
Reference in New Issue
Block a user