9dfa06ffee
Docker image / Build (linux/amd64) (push) Has been cancelled
Docker image / Build (linux/arm64) (push) Has been cancelled
Docker image / Merge release multi-arch manifest (push) Has been cancelled
Docker image / Merge debug multi-arch manifest (push) Has been cancelled
Docker image / Build public push gateway (linux/amd64) (push) Has been cancelled
Docker image / Build public push gateway (linux/arm64) (push) Has been cancelled
Docker image / Publish public push gateway image (push) Has been cancelled
Sprig image / Build (linux/amd64) (push) Has been cancelled
Sprig image / Build (linux/arm64) (push) Has been cancelled
Sprig image / Merge multi-arch manifest (push) Has been cancelled
Harbor Buzz Orchestra / Python tests and lint (push) Has been cancelled
CI / Detect Changed Paths (push) Has been cancelled
CI / Rust Lint (push) Has been cancelled
CI / Unit Tests (push) Has been cancelled
CI / Desktop Core (push) Has been cancelled
CI / Desktop Smoke E2E (1) (push) Has been cancelled
CI / Desktop Smoke E2E (2) (push) Has been cancelled
CI / Desktop Smoke E2E (3) (push) Has been cancelled
CI / Desktop Smoke E2E (4) (push) Has been cancelled
CI / Desktop (push) Has been cancelled
CI / Desktop E2E Relay (push) Has been cancelled
CI / Desktop E2E Integration (1/2) (push) Has been cancelled
CI / Desktop E2E Integration (2/2) (push) Has been cancelled
CI / Desktop E2E Integration (push) Has been cancelled
CI / Backend Integration (relay e2e) (push) Has been cancelled
CI / Relay E2E (push) Has been cancelled
CI / Web (push) Has been cancelled
CI / Mobile (push) Has been cancelled
CI / Security (push) Has been cancelled
CI / Dead Token Reference Guard (push) Has been cancelled
CI / Server Cross-Compile (aarch64-unknown-linux-musl) (push) Has been cancelled
CI / Server Cross-Compile (x86_64-unknown-linux-musl) (push) Has been cancelled
CI / Windows Rust (x86_64-pc-windows-msvc) (push) Has been cancelled
CI / Desktop Build (macOS) (push) Has been cancelled
helm chart / lint + unittest + render matrix (push) Has been cancelled
helm chart / install on kind (gated) (push) Has been cancelled
helm chart / publish chart to GHCR (push) Has been cancelled
Mesh Lifecycle / Relay-Driven Mesh Lifecycle Smoke (push) Has been cancelled
Sprig / Build (aarch64-unknown-linux-musl) (push) Has been cancelled
Sprig / Build (x86_64-unknown-linux-musl) (push) Has been cancelled
Sprig / Publish rolling release (push) Has been cancelled
Sprig / Publish tagged release (push) Has been cancelled
Signed-off-by: cls_宁波本机 <908705107@qq.com>
2974 lines
123 KiB
Rust
2974 lines
123 KiB
Rust
use std::time::Duration;
|
|
|
|
pub const PROTOCOL_VERSION: u32 = 2;
|
|
|
|
/// Reasoning/thinking effort level for providers that support it.
|
|
///
|
|
/// Set via `BUZZ_AGENT_THINKING_EFFORT` (`none|minimal|low|medium|high|xhigh|max`).
|
|
/// When unset the provider's default behaviour is preserved — no thinking
|
|
/// config is sent in the request body.
|
|
///
|
|
/// Provider support (doc-verified, July 2025):
|
|
/// - **Anthropic adaptive**: `low|medium|high|xhigh|max` (model-dependent; see `anthropic_thinking_config`).
|
|
/// `none`/`minimal` are not Anthropic values — rejected at startup.
|
|
/// - **Anthropic manual budget** (claude-3*, opus-4-5): `low|medium|high`; `xhigh`/`max` clamp to high budget.
|
|
/// - **OpenAI Responses / Chat Completions**: effort support is model-dependent and normalized at
|
|
/// request time; `max` is valid for documented max-supporting families such as GPT-5.6.
|
|
/// - **Databricks**: routed by model family (Claude → Anthropic mapping, GPT-5 → Responses, MLflow → Chat).
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
|
pub enum ThinkingEffort {
|
|
None,
|
|
Minimal,
|
|
Low,
|
|
Medium,
|
|
High,
|
|
XHigh,
|
|
Max,
|
|
}
|
|
|
|
impl ThinkingEffort {
|
|
/// Map level to an Anthropic `budget_tokens` value for legacy Claude 3.x / Opus 4.5 models.
|
|
/// `XHigh` and `Max` clamp to the high budget value; the answer-room reserve of 1024 tokens
|
|
/// is applied separately in `anthropic_thinking_config`.
|
|
pub fn anthropic_budget_tokens(self) -> u32 {
|
|
match self {
|
|
ThinkingEffort::Low => 1_024,
|
|
ThinkingEffort::Medium => 8_192,
|
|
ThinkingEffort::High | ThinkingEffort::XHigh | ThinkingEffort::Max => 32_768,
|
|
// None/Minimal are not valid for Anthropic (rejected at startup); treat as zero
|
|
// defensively so a misconfigured call doesn't accidentally enable thinking.
|
|
ThinkingEffort::None | ThinkingEffort::Minimal => 0,
|
|
}
|
|
}
|
|
|
|
/// Map level to an OpenAI `reasoning.effort` / `reasoning_effort` string.
|
|
pub fn openai_effort_str(self) -> &'static str {
|
|
match self {
|
|
ThinkingEffort::None => "none",
|
|
ThinkingEffort::Minimal => "minimal",
|
|
ThinkingEffort::Low => "low",
|
|
ThinkingEffort::Medium => "medium",
|
|
ThinkingEffort::High => "high",
|
|
ThinkingEffort::XHigh => "xhigh",
|
|
ThinkingEffort::Max => "max",
|
|
}
|
|
}
|
|
|
|
/// Map level to an Anthropic `output_config.effort` string.
|
|
/// Returns the level string if supported, or the highest supported level for the model.
|
|
/// Caller must apply model-level clamping via `clamp_for_anthropic_adaptive`.
|
|
pub fn anthropic_effort_str(self) -> &'static str {
|
|
match self {
|
|
ThinkingEffort::Low => "low",
|
|
ThinkingEffort::Medium => "medium",
|
|
ThinkingEffort::High => "high",
|
|
ThinkingEffort::XHigh => "xhigh",
|
|
ThinkingEffort::Max => "max",
|
|
// None/Minimal are rejected at startup for Anthropic; defensive fallback.
|
|
ThinkingEffort::None | ThinkingEffort::Minimal => "low",
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Strip any endpoint-naming prefix from a model name so the family classifiers
|
|
/// (`is_manual_budget_model`, `is_adaptive_thinking_model`, etc.) can match on the canonical
|
|
/// `claude-*` form regardless of how the model is stored in the Databricks catalog.
|
|
///
|
|
/// Rather than maintaining an allowlist of known prefixes, this function finds the first
|
|
/// occurrence of a known model-family token (`claude-`, `gpt-`) and drops everything before
|
|
/// it. This handles any endpoint naming convention without needing to enumerate prefixes.
|
|
///
|
|
/// Examples:
|
|
/// - `databricks-claude-fable-5` → `claude-fable-5`
|
|
/// - `goose-claude-fable-5` → `claude-fable-5`
|
|
/// - `team-x-claude-opus-4-7` → `claude-opus-4-7`
|
|
/// - `goose-gpt-5.5` → `gpt-5.5`
|
|
/// - `llama-3` → `llama-3` (no family token, returned unchanged)
|
|
///
|
|
/// If no family token is present the name is returned unchanged.
|
|
fn strip_catalog_prefix(model: &str) -> &str {
|
|
const FAMILY_TOKENS: &[&str] = &["claude-", "gpt-"];
|
|
let lower = model.to_ascii_lowercase();
|
|
let first_idx = FAMILY_TOKENS.iter().filter_map(|tok| lower.find(tok)).min();
|
|
match first_idx {
|
|
Some(idx) => &model[idx..],
|
|
None => model,
|
|
}
|
|
}
|
|
|
|
/// Build the Anthropic thinking/effort request fields for the given model and effort level.
|
|
///
|
|
/// API shape selection (per Anthropic thinking docs and per-model support table,
|
|
/// https://platform.claude.com/docs/en/build-with-claude/thinking and
|
|
/// https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#supported-models):
|
|
///
|
|
/// **Adaptive families — `thinking:{type:"adaptive"}` activates effort control**:
|
|
///
|
|
/// - Opus 4.6, Opus 4.7, Opus 4.8, Sonnet 4.6: status **Off** — thinking is OFF by default;
|
|
/// `thinking:{type:"adaptive"}` is required to enable thinking; without it no thinking occurs.
|
|
/// - Opus 5, Sonnet 5: status **On** — thinking is on by default (can be disabled);
|
|
/// we still send `thinking:{type:"adaptive"}` so `output_config.effort` is honoured.
|
|
/// - Fable 5, Mythos 5, Mythos Preview: status **Always on** — thinking cannot be disabled;
|
|
/// we still send `thinking:{type:"adaptive"}` so `output_config.effort` is honoured.
|
|
///
|
|
/// In all three sub-buckets `output_config: {effort}` controls depth, clamped per-model.
|
|
/// Also sends `thinking: {display:"summarized"}` so thinking text is always visible in the
|
|
/// observer feed (without this, Anthropic defaults to `display:"omitted"` on newest models).
|
|
///
|
|
/// **Manual-budget families** — `thinking: {type:"enabled", budget_tokens}`.
|
|
/// `budget_tokens` is clamped to `min(level_budget, max_output_tokens - 1024)` to preserve
|
|
/// at least 1024 answer tokens. If the result is < 1024 (i.e., `max_output_tokens <= 2047`),
|
|
/// thinking is omitted entirely with a `warn!`.
|
|
/// Doc-verified: claude-3* (legacy), claude-opus-4-5 (effort page: "uses manual thinking").
|
|
/// Also sends `display:"summarized"` to ensure thinking text is returned.
|
|
///
|
|
/// **Everything else** — omit both fields. This includes unknown/future `claude-*` names
|
|
/// not yet in the support table. Safer to omit than to guess an unverified shape.
|
|
///
|
|
/// The Databricks `databricks-` and other endpoint-naming prefixes are stripped before
|
|
/// matching so that `databricks-claude-opus-4-7`, `goose-claude-fable-5`, and
|
|
/// `team-x-claude-opus-4-7` all route to the correct bucket. See `strip_catalog_prefix`.
|
|
///
|
|
/// Returns `(thinking_field, output_config_field)` where each is `None` if not applicable.
|
|
pub fn anthropic_thinking_config(
|
|
effective_model: &str,
|
|
effort: ThinkingEffort,
|
|
max_output_tokens: u32,
|
|
) -> (Option<serde_json::Value>, Option<serde_json::Value>) {
|
|
use serde_json::json;
|
|
// Normalise the model name for matching: strip any endpoint-naming prefix
|
|
// (e.g. "databricks-claude-opus-4-7" → "claude-opus-4-7",
|
|
// "goose-claude-fable-5" → "claude-fable-5",
|
|
// "team-x-claude-opus-4-7" → "claude-opus-4-7").
|
|
let model = strip_catalog_prefix(effective_model);
|
|
|
|
if is_manual_budget_model(model) {
|
|
// Manual-budget shape: budget_tokens must be strictly < max_tokens AND must leave
|
|
// at least MIN_ANSWER_TOKENS (1024) for the visible answer. The Anthropic API
|
|
// requires budget_tokens < max_tokens AND budget_tokens >= 1024.
|
|
//
|
|
// Clamp: budget = min(level_budget, max_output_tokens - MIN_ANSWER_TOKENS).
|
|
// If result < MIN_ANSWER_TOKENS, thinking would starve the answer — omit thinking
|
|
// entirely and warn instead of emitting an invalid or answer-starving budget.
|
|
const MIN_ANSWER_TOKENS: u32 = 1024;
|
|
let level_budget = effort.anthropic_budget_tokens();
|
|
let headroom = max_output_tokens.saturating_sub(MIN_ANSWER_TOKENS);
|
|
let budget = level_budget.min(headroom);
|
|
if budget < MIN_ANSWER_TOKENS {
|
|
tracing::warn!(
|
|
max_output_tokens,
|
|
level_budget,
|
|
headroom,
|
|
"BUZZ_AGENT_THINKING_EFFORT: max_output_tokens too small to fit thinking budget + answer headroom; omitting thinking fields"
|
|
);
|
|
return (None, None);
|
|
}
|
|
(
|
|
Some(json!({ "type": "enabled", "budget_tokens": budget, "display": "summarized" })),
|
|
None,
|
|
)
|
|
} else if is_adaptive_thinking_model(model) {
|
|
// Adaptive families: we always send type:"adaptive" to activate output_config.effort.
|
|
// Sub-bucket A (Off: Opus 4.6/4.7/4.8, Sonnet 4.6): this field is required to enable
|
|
// thinking at all. Sub-bucket B (On: Opus 5/Sonnet 5) and sub-bucket C (Always on:
|
|
// Fable 5/Mythos 5/Mythos Preview): thinking is already on; we send the field so
|
|
// output_config.effort is honoured, not to enable thinking.
|
|
// Apply per-model effort clamping: if the requested level exceeds the model's
|
|
// doc-verified maximum, clamp down to the highest supported level with a warning.
|
|
let clamped = clamp_adaptive_effort(model, effort);
|
|
(
|
|
Some(json!({ "type": "adaptive", "display": "summarized" })),
|
|
Some(json!({ "effort": clamped.anthropic_effort_str() })),
|
|
)
|
|
} else {
|
|
// Unrecognised or unverified model name — omit both fields rather than guess.
|
|
// This includes unknown future claude-* names not yet in the support table.
|
|
(None, None)
|
|
}
|
|
}
|
|
|
|
/// Returns true for adaptive Anthropic models that support the `xhigh` effort level.
|
|
///
|
|
/// Used by both `clamp_adaptive_effort` (request-time) and `anthropic_efforts_for_model`
|
|
/// (UI capability table) to keep xhigh-support classification in a single place.
|
|
///
|
|
/// `model` must already have catalog prefixes stripped (via `strip_catalog_prefix`).
|
|
fn anthropic_model_supports_xhigh(model: &str) -> bool {
|
|
model.starts_with("claude-opus-4-7")
|
|
|| model.starts_with("claude-opus-4-8")
|
|
|| model.starts_with("claude-opus-5")
|
|
|| model.starts_with("claude-sonnet-5")
|
|
|| model.starts_with("claude-fable-5")
|
|
|| model.starts_with("claude-mythos-5")
|
|
}
|
|
|
|
/// Clamp the requested effort level to the highest doc-verified level for the given adaptive model.
|
|
///
|
|
/// Doc-verified availability (Anthropic effort page, July 2025):
|
|
/// - `max`: Opus 4.8, 4.7, 4.6; Sonnet 5.x, 4.6; Fable 5; Mythos 5; Mythos Preview
|
|
/// - `xhigh`: Opus 4.8, 4.7; Sonnet 5.x; Fable 5; Mythos 5
|
|
/// (NOT Opus 4.6, Sonnet 4.6, or Mythos Preview)
|
|
/// - `low|medium|high`: all adaptive families
|
|
///
|
|
/// If the requested level is not available for the model, clamps down to the highest
|
|
/// supported level below the requested one, and logs a warning. This is dynamic (not
|
|
/// startup-time) because `session/set_model` can change the model after startup.
|
|
///
|
|
/// `model` must already have catalog prefixes stripped (via `strip_catalog_prefix`).
|
|
pub fn clamp_adaptive_effort(model: &str, effort: ThinkingEffort) -> ThinkingEffort {
|
|
// Models that support all levels including xhigh (and max).
|
|
let supports_xhigh = anthropic_model_supports_xhigh(model);
|
|
|
|
let clamped = if supports_xhigh {
|
|
effort // all levels pass through
|
|
} else if effort == ThinkingEffort::XHigh {
|
|
// xhigh not available for this model; clamp to high (the highest supported below xhigh).
|
|
ThinkingEffort::High
|
|
} else {
|
|
effort // low/medium/high/max all pass through for the other adaptive families
|
|
};
|
|
|
|
if clamped != effort {
|
|
tracing::warn!(
|
|
model,
|
|
requested = effort.openai_effort_str(),
|
|
clamped = clamped.openai_effort_str(),
|
|
"BUZZ_AGENT_THINKING_EFFORT is not available for this model; clamping to highest supported level"
|
|
);
|
|
}
|
|
clamped
|
|
}
|
|
|
|
/// Returns true if `lower_model` contains `token` as a bounded family segment — i.e., the
|
|
/// token is immediately followed by end-of-string or a `-` separator (not a digit or letter).
|
|
///
|
|
/// This prevents:
|
|
/// - `gpt-5.1` from matching `gpt-5.10` (digit follows the `1`)
|
|
/// - `gpt-5-1` from matching `gpt-5-1106` (digit follows the `1`)
|
|
/// - `gpt-5-4` from matching `gpt-5-4o` (letter follows the `4`)
|
|
///
|
|
/// Gateway prefixes (`databricks-`) and date/build suffixes (`-2025-04-01`) are allowed
|
|
/// because they start with `-` which is the only permitted boundary character.
|
|
fn gpt5_token_matches(lower_model: &str, token: &str) -> bool {
|
|
let mut start = 0;
|
|
while let Some(pos) = lower_model[start..].find(token) {
|
|
let abs = start + pos;
|
|
let after = abs + token.len();
|
|
// The character immediately after the token must be end-of-string or '-'.
|
|
// Any alphanumeric character (digit OR letter) means this is a longer token, not
|
|
// the family we're looking for.
|
|
let safe_suffix = lower_model[after..].chars().next().is_none_or(|c| c == '-');
|
|
if safe_suffix {
|
|
return true;
|
|
}
|
|
start = abs + 1;
|
|
}
|
|
false
|
|
}
|
|
|
|
/// Like `gpt5_token_matches` but additionally rejects short version-like numeric suffixes —
|
|
/// used for the base `gpt-5` / `gpt5` token to avoid false-matching unrecognized versions.
|
|
///
|
|
/// After a `-` separator:
|
|
/// - `-<non-digit>…` e.g. `-pro` → **accepted** (capability suffix, no digits)
|
|
/// - `digit_run == 1-3` AND the char right after the digits is a **letter** e.g. `-4o` →
|
|
/// **accepted** (real variant shape: digit + letter)
|
|
/// - `digit_run == 1-3` AND the char after the digits is end-of-string, `-`, `.`, or other
|
|
/// separator e.g. `-10`, `-10-preview` → **rejected** (version-like suffix)
|
|
/// - `digit_run >= 4` regardless of what follows e.g. `-1106`, `-1106-preview`, `-0514` →
|
|
/// **accepted** (date/build segment)
|
|
fn gpt5_base_matches(lower_model: &str, token: &str) -> bool {
|
|
let mut start = 0;
|
|
while let Some(pos) = lower_model[start..].find(token) {
|
|
let abs = start + pos;
|
|
let after = abs + token.len();
|
|
let rest = &lower_model[after..];
|
|
let safe_suffix = if rest.is_empty() {
|
|
// End of string — clean boundary.
|
|
true
|
|
} else if let Some(tail) = rest.strip_prefix('-') {
|
|
// Count leading digits in the suffix component.
|
|
let digit_run: usize = tail.chars().take_while(|c| c.is_ascii_digit()).count();
|
|
if digit_run == 0 {
|
|
// No leading digit (e.g. '-pro'): capability suffix → accepted.
|
|
true
|
|
} else if digit_run >= 4 {
|
|
// 4+ digit run (e.g. '-1106', '-1106-preview', '-0514'): date/build → accepted.
|
|
true
|
|
} else {
|
|
// 1-3 digit run: accepted only if the char right after the digits is a letter
|
|
// (real variant shape like '-4o'). Separator/EOS after short digits is
|
|
// version-like (e.g. '-10', '-10-preview') → rejected.
|
|
tail[digit_run..]
|
|
.chars()
|
|
.next()
|
|
.is_some_and(|c| c.is_ascii_alphabetic())
|
|
}
|
|
} else {
|
|
// Dot, letter, or other non-hyphen character directly after token → not base.
|
|
false
|
|
};
|
|
if safe_suffix {
|
|
return true;
|
|
}
|
|
start = abs + 1;
|
|
}
|
|
false
|
|
}
|
|
|
|
/// Returns the set of `reasoning.effort` values supported by a given OpenAI model family.
|
|
///
|
|
/// Doc-verified availability (OpenAI model pages, July 2025):
|
|
///
|
|
/// | Model | Supported effort values |
|
|
/// |-------------|-------------------------------------------|
|
|
/// | gpt-5-pro | `high` only |
|
|
/// | gpt-5.6 | `none, low, medium, high, xhigh, max` |
|
|
/// | gpt-5.5 | `none, low, medium, high, xhigh` |
|
|
/// | gpt-5.4 | `none, low, medium, high, xhigh` |
|
|
/// | gpt-5.1 | `none, low, medium, high` |
|
|
/// | gpt-5 (base)| `minimal, low, medium, high` |
|
|
/// | unknown | not doc-verified — `max` clamps to `xhigh` |
|
|
///
|
|
/// Note the `none` vs `minimal` split: `gpt-5` (base) supports `minimal` but not `none`;
|
|
/// `gpt-5.1`/`gpt-5.4`/`gpt-5.5`/`gpt-5.6` support `none` but not `minimal`. These are matched via
|
|
/// nearest-supported fallback in `normalize_effort_for_openai_route`.
|
|
///
|
|
/// Match order: `-pro` variant checked before versioned strings to prevent `gpt-5-pro` from
|
|
/// falling into the `gpt-5` base bucket (substring "gpt-5" is shared).
|
|
///
|
|
/// `model` is a raw model name (may include Databricks gateway prefixes or date suffixes).
|
|
/// Unknown models return `None` — callers pass through values except `max`, which clamps to
|
|
/// `xhigh` until support is confirmed.
|
|
/// Versioned tokens use `gpt5_token_matches` (end-of-string or `-` boundary, blocking digit
|
|
/// and letter continuations). The base token uses `gpt5_base_matches`, which additionally
|
|
/// rejects short `-<1-3 digit>` suffixes that look like two-digit version numbers.
|
|
fn openai_efforts_for_model(model: &str) -> Option<&'static [ThinkingEffort]> {
|
|
// Effort ordered from lowest to highest for each family.
|
|
const GPT5_PRO: &[ThinkingEffort] = &[ThinkingEffort::High];
|
|
const GPT5_6: &[ThinkingEffort] = &[
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
];
|
|
const GPT5_5_AND_5_4: &[ThinkingEffort] = &[
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
];
|
|
const GPT5_1: &[ThinkingEffort] = &[
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
];
|
|
const GPT5_BASE: &[ThinkingEffort] = &[
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
];
|
|
|
|
let lower = model.to_ascii_lowercase();
|
|
// Check gpt-5-pro before gpt-5.5 / gpt-5.4 etc. to avoid the `-pro` name
|
|
// matching the base "gpt-5" prefix first.
|
|
if gpt5_token_matches(&lower, "gpt-5-pro") || gpt5_token_matches(&lower, "gpt5-pro") {
|
|
Some(GPT5_PRO)
|
|
} else if gpt5_token_matches(&lower, "gpt-5.6")
|
|
|| gpt5_token_matches(&lower, "gpt5.6")
|
|
|| gpt5_token_matches(&lower, "gpt-5-6")
|
|
|| gpt5_token_matches(&lower, "gpt5-6")
|
|
{
|
|
Some(GPT5_6)
|
|
} else if gpt5_token_matches(&lower, "gpt-5.5")
|
|
|| gpt5_token_matches(&lower, "gpt5.5")
|
|
|| gpt5_token_matches(&lower, "gpt-5-5")
|
|
|| gpt5_token_matches(&lower, "gpt5-5")
|
|
|| gpt5_token_matches(&lower, "gpt-5.4")
|
|
|| gpt5_token_matches(&lower, "gpt5.4")
|
|
|| gpt5_token_matches(&lower, "gpt-5-4")
|
|
|| gpt5_token_matches(&lower, "gpt5-4")
|
|
{
|
|
// gpt-5.5 and gpt-5.4 share the same effort availability table.
|
|
Some(GPT5_5_AND_5_4)
|
|
} else if gpt5_token_matches(&lower, "gpt-5.1")
|
|
|| gpt5_token_matches(&lower, "gpt5.1")
|
|
|| gpt5_token_matches(&lower, "gpt-5-1")
|
|
|| gpt5_token_matches(&lower, "gpt5-1")
|
|
{
|
|
Some(GPT5_1)
|
|
} else if gpt5_base_matches(&lower, "gpt-5") || gpt5_base_matches(&lower, "gpt5") {
|
|
// Base gpt-5 (no version suffix matching any of the above).
|
|
Some(GPT5_BASE)
|
|
} else {
|
|
// Unknown model — not doc-verified; server validates.
|
|
None
|
|
}
|
|
}
|
|
|
|
/// Returns the effort capability set for a given Anthropic model.
|
|
///
|
|
/// This is the single production source of truth for Anthropic family routing.
|
|
/// Both `anthropic_thinking_config` (request-time) and the effort-table UI
|
|
/// (`valid_effort_values_for_provider_model`, via its Anthropic branch) must
|
|
/// derive their behaviour from this helper so the two stay in sync.
|
|
///
|
|
/// Returns `(valid_values, default)` where:
|
|
/// - `valid_values` is the static slice of `ThinkingEffort` values accepted
|
|
/// by this model family's effort dropdown.
|
|
/// - `default` is `None` for manual-budget models (no semantic default —
|
|
/// user must choose) or `Some(High)` for adaptive families.
|
|
///
|
|
/// `model` must already have catalog prefixes stripped (via `strip_catalog_prefix`).
|
|
pub fn anthropic_efforts_for_model(
|
|
model: &str,
|
|
) -> (&'static [ThinkingEffort], Option<ThinkingEffort>) {
|
|
const MANUAL: &[ThinkingEffort] = &[
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
];
|
|
const ADAPTIVE_XHIGH: &[ThinkingEffort] = &[
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
];
|
|
const ADAPTIVE_NO_XHIGH: &[ThinkingEffort] = &[
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::Max,
|
|
];
|
|
|
|
if is_manual_budget_model(model) {
|
|
return (MANUAL, None);
|
|
}
|
|
if is_adaptive_thinking_model(model) {
|
|
// Reuse `anthropic_model_supports_xhigh` (the single source of truth
|
|
// shared with `clamp_adaptive_effort`) — no side-effects, no duplication.
|
|
if anthropic_model_supports_xhigh(model) {
|
|
return (ADAPTIVE_XHIGH, Some(ThinkingEffort::High));
|
|
} else {
|
|
return (ADAPTIVE_NO_XHIGH, Some(ThinkingEffort::High));
|
|
}
|
|
}
|
|
// Unknown Anthropic model — assume full adaptive (xhigh-capable) as a safe default.
|
|
(ADAPTIVE_XHIGH, Some(ThinkingEffort::High))
|
|
}
|
|
|
|
/// Resolve the nearest supported effort level for a given OpenAI model.
|
|
///
|
|
/// When the requested effort is not in the model's supported set, falls back to the
|
|
/// nearest supported level using this preference order:
|
|
///
|
|
/// - `none` ↔ `minimal` are each other's first fallback (the none/minimal split across
|
|
/// model families means the "closest analogue" is the other form before jumping to `low`).
|
|
/// - Above that pair: upward clamp first, then downward (prefer more thinking over less).
|
|
/// - `xhigh` falls back to `high` when not supported (no model skips from `high` to `xhigh`).
|
|
/// - `max` passes through for model families whose table includes it; otherwise it resolves to
|
|
/// the nearest supported level.
|
|
///
|
|
/// Logs a `warn!` on every substitution.
|
|
fn resolve_openai_effort(
|
|
model: &str,
|
|
requested: ThinkingEffort,
|
|
supported: &[ThinkingEffort],
|
|
) -> ThinkingEffort {
|
|
if supported.contains(&requested) {
|
|
return requested;
|
|
}
|
|
|
|
// Build a candidate list ordered by preference: the "other" form of none/minimal first,
|
|
// then the levels sorted nearest to requested (ascending distance).
|
|
let candidates: Vec<ThinkingEffort> = {
|
|
// none ↔ minimal are each other's first fallback.
|
|
let peer = match requested {
|
|
ThinkingEffort::None => Some(ThinkingEffort::Minimal),
|
|
ThinkingEffort::Minimal => Some(ThinkingEffort::None),
|
|
_ => None,
|
|
};
|
|
// All supported values sorted by distance (abs diff in ordinal), upward ties win.
|
|
let mut by_dist: Vec<ThinkingEffort> = supported.to_vec();
|
|
by_dist.sort_by_key(|&e| {
|
|
let dist = (e as i32 - requested as i32).unsigned_abs();
|
|
// Prefer upward (e > requested) to break ties between equidistant values.
|
|
let up = if e >= requested { 0u32 } else { 1 };
|
|
(dist, up)
|
|
});
|
|
// Peer first, then by distance.
|
|
let mut result = Vec::new();
|
|
if let Some(p) = peer {
|
|
if supported.contains(&p) {
|
|
result.push(p);
|
|
}
|
|
}
|
|
for e in by_dist {
|
|
if !result.contains(&e) {
|
|
result.push(e);
|
|
}
|
|
}
|
|
result
|
|
};
|
|
|
|
let resolved = candidates
|
|
.into_iter()
|
|
.next()
|
|
.expect("supported is non-empty");
|
|
|
|
tracing::warn!(
|
|
%model,
|
|
requested = requested.openai_effort_str(),
|
|
resolved = resolved.openai_effort_str(),
|
|
"BUZZ_AGENT_THINKING_EFFORT={} is not supported by this OpenAI model; using nearest supported level",
|
|
requested.openai_effort_str(),
|
|
);
|
|
resolved
|
|
}
|
|
|
|
/// Normalize the effort value for an OpenAI-shaped request body (Chat Completions or Responses).
|
|
///
|
|
/// Per-model effort availability is applied for doc-verified OpenAI model families. A requested
|
|
/// level not in the model's supported set is substituted with the nearest supported level (see
|
|
/// `resolve_openai_effort` for preference order). For unknown/unverified models, `max` is clamped
|
|
/// to `xhigh` because its support cannot be confirmed; all other values pass through unchanged.
|
|
///
|
|
/// Applies to pure-OpenAI request paths AND DBv2 OpenAI-shaped routes.
|
|
///
|
|
/// Doc-verified model table (July 2025):
|
|
/// - `gpt-5-pro`: `high` only
|
|
/// - `gpt-5.6`: `none, low, medium, high, xhigh, max`
|
|
/// - `gpt-5.5`, `gpt-5.4`: `none, low, medium, high, xhigh`
|
|
/// - `gpt-5.1`: `none, low, medium, high`
|
|
/// - `gpt-5` (base): `minimal, low, medium, high`
|
|
/// - unknown: `max` clamps to `xhigh`; other values pass through
|
|
pub fn normalize_effort_for_openai_route(effort: ThinkingEffort, model: &str) -> ThinkingEffort {
|
|
match openai_efforts_for_model(model) {
|
|
Some(supported) => resolve_openai_effort(model, effort, supported),
|
|
None if effort == ThinkingEffort::Max => {
|
|
tracing::warn!(
|
|
requested = "max",
|
|
resolved = "xhigh",
|
|
"BUZZ_AGENT_THINKING_EFFORT=max not confirmed for unknown OpenAI model; clamping to xhigh"
|
|
);
|
|
ThinkingEffort::XHigh
|
|
}
|
|
None => effort,
|
|
}
|
|
}
|
|
|
|
/// Normalize the effort value for an Anthropic-shaped request body (Messages API).
|
|
///
|
|
/// Anthropic-shaped bodies (`anthropic_body`) do not have a `none` or `minimal` concept —
|
|
/// the thinking block is either present (with a level) or absent. When `none` or `minimal`
|
|
/// is configured, we omit the thinking fields entirely and log a warning (omission = provider
|
|
/// default; default-on/always-on adaptive models may still think). This handles `DatabricksV2`
|
|
/// sessions where the route can switch from GPT to Claude via `session/set_model` after startup.
|
|
///
|
|
/// Returns `None` to signal "omit thinking fields", or the original effort if it is a valid
|
|
/// Anthropic level.
|
|
pub fn normalize_effort_for_anthropic_route(effort: ThinkingEffort) -> Option<ThinkingEffort> {
|
|
match effort {
|
|
ThinkingEffort::None | ThinkingEffort::Minimal => {
|
|
tracing::warn!(
|
|
requested = effort.openai_effort_str(),
|
|
"BUZZ_AGENT_THINKING_EFFORT={} is not expressible as an Anthropic thinking level; \
|
|
omitting thinking fields (provider default; default-on/always-on adaptive models may still think)",
|
|
effort.openai_effort_str()
|
|
);
|
|
None
|
|
}
|
|
other => Some(other),
|
|
}
|
|
}
|
|
|
|
/// Returns true for Claude model families that use manual thinking budgets (doc-verified, July 2025).
|
|
///
|
|
/// Source: https://platform.claude.com/docs/en/build-with-claude/extended-thinking (support table)
|
|
/// - claude-3*: legacy manual budget (all Claude 3.x variants).
|
|
/// - claude-opus-4-5: effort page states "uses manual thinking, where effort works alongside
|
|
/// the thinking token budget" — manual bucket, not adaptive.
|
|
///
|
|
/// `model` must already have catalog prefixes stripped (via `strip_catalog_prefix`).
|
|
fn is_manual_budget_model(model: &str) -> bool {
|
|
model.starts_with("claude-3") || model == "claude-opus-4-5"
|
|
}
|
|
|
|
/// Returns true for Claude model families that use adaptive thinking (doc-verified against
|
|
/// https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#supported-models).
|
|
///
|
|
/// **Sub-bucket A — status Off (thinking OFF until `thinking:{type:"adaptive"}` is sent)**:
|
|
/// Opus 4.6, Opus 4.7, Opus 4.8, Sonnet 4.6.
|
|
///
|
|
/// **Sub-bucket B — status On (thinking on by default; can be disabled)**:
|
|
/// Opus 5, Sonnet 5.
|
|
/// We still send `thinking:{type:"adaptive"}` so `output_config.effort` is honoured.
|
|
///
|
|
/// **Sub-bucket C — status Always on (thinking cannot be disabled)**:
|
|
/// Fable 5, Mythos 5, Mythos Preview.
|
|
/// We still send `thinking:{type:"adaptive"}` so `output_config.effort` is honoured.
|
|
///
|
|
/// All three sub-buckets accept the same request shape. The distinction matters only when
|
|
/// thinking effort is NOT configured: sub-bucket B/C models still produce thinking even
|
|
/// without us sending the field; sub-bucket A models do not.
|
|
///
|
|
/// Note: Opus 4.5 is NOT in this bucket — it uses manual budget (see `is_manual_budget_model`).
|
|
/// No prefix wildcards over version numbers; each entry is doc-verified explicitly.
|
|
///
|
|
/// `model` must already have catalog prefixes stripped (via `strip_catalog_prefix`).
|
|
fn is_adaptive_thinking_model(model: &str) -> bool {
|
|
// Exact version strings for Opus 4.x adaptive models (4.6, 4.7, 4.8).
|
|
// Opus 4.5 is excluded — manual budget only.
|
|
model.starts_with("claude-opus-4-6")
|
|
|| model.starts_with("claude-opus-4-7")
|
|
|| model.starts_with("claude-opus-4-8")
|
|
|| model.starts_with("claude-opus-5")
|
|
// Sonnet 5.x (any patch/date suffix after "claude-sonnet-5").
|
|
|| model.starts_with("claude-sonnet-5")
|
|
// Sonnet 4.6 exactly (not Sonnet 4.5 or earlier — not in the adaptive table).
|
|
|| model.starts_with("claude-sonnet-4-6")
|
|
// Fable 5 and Mythos 5 (Always on — thinking cannot be disabled, July 2025).
|
|
|| model.starts_with("claude-fable-5")
|
|
|| model.starts_with("claude-mythos-5")
|
|
// Mythos Preview (Always on — thinking cannot be disabled, July 2025).
|
|
// Note: xhigh is NOT available on Mythos Preview — clamp_adaptive_effort handles this.
|
|
|| model.starts_with("claude-mythos-preview")
|
|
}
|
|
|
|
/// Reasoning summary mode for the OpenAI Responses API route.
|
|
///
|
|
/// Controls the `reasoning.summary` field sent alongside `reasoning.effort` in
|
|
/// `responses_body`. The Responses API only returns populated `summary` arrays
|
|
/// when a summary mode is requested — without it, `summary: []` is returned and
|
|
/// the observer feed shows no reasoning text even though the model billed thinking
|
|
/// tokens.
|
|
///
|
|
/// **Responses-route only.** On the Anthropic route, thinking blocks contain the
|
|
/// full reasoning text directly (no summary concept); this field is ignored there.
|
|
/// On Chat Completions and OpenRouter paths the field is also ignored.
|
|
///
|
|
/// Set via `BUZZ_AGENT_THINKING_SUMMARY` (`auto|concise|detailed`).
|
|
/// Unset/empty → `auto` (the provider chooses the best available summary for the
|
|
/// model). Use `detailed` for maximum reasoning visibility in the observer feed.
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
pub enum ThinkingSummary {
|
|
/// Provider selects the best available summary format for the model.
|
|
Auto,
|
|
/// Shorter summaries — lower token overhead.
|
|
Concise,
|
|
/// Full-length summaries — maximum reasoning visibility.
|
|
Detailed,
|
|
}
|
|
|
|
impl ThinkingSummary {
|
|
/// The string value sent in the `reasoning.summary` field.
|
|
pub fn as_str(self) -> &'static str {
|
|
match self {
|
|
ThinkingSummary::Auto => "auto",
|
|
ThinkingSummary::Concise => "concise",
|
|
ThinkingSummary::Detailed => "detailed",
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Parse `BUZZ_AGENT_THINKING_SUMMARY`. Pure (env-free) for testability.
|
|
///
|
|
/// Unset or empty → `Auto` (the safe default that works for all Responses-capable models).
|
|
/// Invalid value → startup error.
|
|
pub fn parse_thinking_summary(raw: Option<&str>) -> Result<ThinkingSummary, String> {
|
|
match raw.map(|s| s.trim().to_ascii_lowercase()).as_deref() {
|
|
None | Some("") => Ok(ThinkingSummary::Auto),
|
|
Some("auto") => Ok(ThinkingSummary::Auto),
|
|
Some("concise") => Ok(ThinkingSummary::Concise),
|
|
Some("detailed") => Ok(ThinkingSummary::Detailed),
|
|
Some(other) => Err(format!(
|
|
"config: BUZZ_AGENT_THINKING_SUMMARY={other} not supported (use auto|concise|detailed)"
|
|
)),
|
|
}
|
|
}
|
|
|
|
/// Parse `BUZZ_AGENT_THINKING_EFFORT`. Pure (env-free) for testability.
|
|
pub fn parse_thinking_effort(raw: Option<&str>) -> Result<Option<ThinkingEffort>, String> {
|
|
match raw.map(|s| s.trim().to_ascii_lowercase()).as_deref() {
|
|
None | Some("") => Ok(None),
|
|
Some("none") => Ok(Some(ThinkingEffort::None)),
|
|
Some("minimal") => Ok(Some(ThinkingEffort::Minimal)),
|
|
Some("low") => Ok(Some(ThinkingEffort::Low)),
|
|
Some("medium") => Ok(Some(ThinkingEffort::Medium)),
|
|
Some("high") => Ok(Some(ThinkingEffort::High)),
|
|
Some("xhigh") => Ok(Some(ThinkingEffort::XHigh)),
|
|
Some("max") => Ok(Some(ThinkingEffort::Max)),
|
|
Some(other) => Err(format!(
|
|
"config: BUZZ_AGENT_THINKING_EFFORT={other} not supported (use none|minimal|low|medium|high|xhigh|max)"
|
|
)),
|
|
}
|
|
}
|
|
|
|
pub const MAX_PROMPT_BYTES: usize = 1024 * 1024;
|
|
pub const MAX_SYSTEM_PROMPT_BYTES: usize = 512 * 1024;
|
|
/// Total per-result byte ceiling (text + images). Sized for image-bearing
|
|
/// results — view_image can legitimately return multi-MiB base64 payloads.
|
|
/// Text is governed by the much smaller `BUZZ_AGENT_MAX_TOOL_RESULT_TEXT_BYTES`.
|
|
pub const MAX_TOOL_RESULT_BYTES: usize = 8 * 1024 * 1024;
|
|
/// Default cap on the *text* portion of a single tool result. Oversized text
|
|
/// is middle-elided before it enters history; without this, one fat `cat`
|
|
/// burns the context window and forces a lossy handoff. 50 KiB matches the
|
|
/// shell-output caps in sprout-dev-mcp, goose, and pi; codex defaults to
|
|
/// 10 KB. Tunable via `BUZZ_AGENT_MAX_TOOL_RESULT_TEXT_BYTES`.
|
|
pub const DEFAULT_TOOL_RESULT_TEXT_BYTES: usize = 50 * 1024;
|
|
pub const MAX_TOOL_CALLS_PER_TURN: usize = 64;
|
|
|
|
pub const HANDOFF_MAX_OUTPUT_TOKENS: u32 = 8192;
|
|
|
|
pub const HANDOFF_ORIGINAL_TASK_MAX_BYTES: usize = 16 * 1024;
|
|
|
|
pub const HANDOFF_MAX_TOOL_NAMES: usize = 20;
|
|
|
|
/// Maximum reactive context-recovery attempts per `run()`. A provider
|
|
/// context-window 400 is recoverable — shrink history and retry — but the
|
|
/// retry must be bounded: `max_rounds` defaults to `0` (unbounded), so without
|
|
/// its own budget a request that stays oversized after every rescue would
|
|
/// retry forever. On exhaustion the error surfaces to the caller, which is a
|
|
/// visible failure rather than a silent infinite rescue.
|
|
pub const MAX_CONTEXT_RECOVERIES_PER_RUN: u32 = 3;
|
|
|
|
/// Floor for the reactive handoff's history-prompt budget, in bytes. Each
|
|
/// recovery attempt halves the budget so the rescue summarize call can escape
|
|
/// an overstated `max_context_tokens`, but halving must terminate: below this
|
|
/// the prompt can no longer carry a useful summary, so the recovery gives up
|
|
/// and surfaces the error instead of issuing ever-smaller doomed requests.
|
|
pub const HANDOFF_MIN_PROMPT_BUDGET_BYTES: usize = 4 * 1024;
|
|
|
|
const DEFAULT_SYSTEM_PROMPT: &str =
|
|
"You are buzz-agent. Use the provided tools to act. Tool calls are your only output.";
|
|
|
|
#[derive(Debug, Clone, Copy, PartialEq)]
|
|
pub enum Provider {
|
|
Anthropic,
|
|
OpenAi,
|
|
/// Databricks model serving. Routes to `{base_url}/serving-endpoints/{model}/invocations`
|
|
/// with a dynamically-acquired bearer (OAuth 2.0 PKCE, or static `DATABRICKS_TOKEN`).
|
|
/// Wire format is OpenAI-chat-compatible — reuses the same body builder and parser.
|
|
Databricks,
|
|
/// Databricks AI Gateway v2. Routes by model family through the gateway's
|
|
/// OpenAI Responses, Anthropic Messages, or MLflow Chat Completions paths.
|
|
DatabricksV2,
|
|
/// OpenRouter multi-provider gateway. Routes to `{base_url}/chat/completions` with bearer auth. Wire format is OpenAI-chat-compatible.
|
|
OpenRouter,
|
|
}
|
|
|
|
/// Which OpenAI-family HTTP API to call. Set via `OPENAI_COMPAT_API`
|
|
/// (`auto|chat|responses`); ignored when `provider = Anthropic`. `Auto`
|
|
/// picks Responses for `*.openai.com`, Chat Completions otherwise, and
|
|
/// permits a one-shot chat→responses upgrade on a "use /v1/responses"
|
|
/// provider error.
|
|
#[derive(Debug, Clone, Copy, PartialEq)]
|
|
pub enum OpenAiApi {
|
|
Chat,
|
|
Responses,
|
|
Auto,
|
|
}
|
|
|
|
#[derive(Debug, Clone)]
|
|
pub struct Config {
|
|
pub provider: Provider,
|
|
pub system_prompt: String,
|
|
pub max_rounds: u32,
|
|
pub max_output_tokens: u32,
|
|
pub llm_timeout: Duration,
|
|
pub tool_timeout: Duration,
|
|
pub mcp_init_timeout: Duration,
|
|
pub mcp_max_restart_attempts: u32,
|
|
pub mcp_restart_base_ms: u64,
|
|
pub mcp_restart_max_ms: u64,
|
|
pub max_sessions: usize,
|
|
pub max_line_bytes: usize,
|
|
pub max_history_bytes: usize,
|
|
/// Per-tool-result cap on text content. Oversized text is middle-elided
|
|
/// (head + tail kept) before entering history. Images are exempt — they
|
|
/// are bounded by [`MAX_TOOL_RESULT_BYTES`] and accounted separately.
|
|
/// Set via `BUZZ_AGENT_MAX_TOOL_RESULT_TEXT_BYTES`.
|
|
pub max_tool_result_text_bytes: usize,
|
|
/// Provider context window in tokens used to gate handoff. The handoff
|
|
/// fires when the previous request's (cache-summed) input tokens cross the
|
|
/// handoff threshold for this budget, before the next request can exceed
|
|
/// the window and 400. Default 200_000 — matching Claude 4.x windows;
|
|
/// operators lower/raise it for other models. Set via
|
|
/// `BUZZ_AGENT_MAX_CONTEXT_TOKENS`.
|
|
pub max_context_tokens: u64,
|
|
/// Maximum context-handoff attempts permitted within a single
|
|
/// `session/prompt` turn. Caps runaway compaction loops inside one turn;
|
|
/// does NOT limit handoffs across a session's lifetime — a long-lived
|
|
/// session can compact on every successive turn without hitting this bound.
|
|
/// Set via `BUZZ_AGENT_MAX_HANDOFFS`. Default 10.
|
|
pub max_handoffs: usize,
|
|
pub max_parallel_tools: usize,
|
|
pub hook_timeout: Duration,
|
|
/// Maximum `_Stop` rejections per prompt. Default 3. Set to 0 to
|
|
/// disable `_Stop` hooks entirely (agent always honors end_turn).
|
|
pub stop_max_rejections: u32,
|
|
/// Remind the model to publish when a turn is about to end without any
|
|
/// recognized attempt to post to Buzz. Default off; opt in per agent with
|
|
/// `BUZZ_AGENT_REQUIRE_REPLY=1`.
|
|
///
|
|
/// Advisory only: at most `MAX_REPLY_NAGS` reminders (see `agent.rs`),
|
|
/// then the turn ends regardless. Bounded by the same
|
|
/// `stop_max_rejections` budget as `_Stop` hooks, which is the outer cap on
|
|
/// all end-turn objections — at the default 3 both reminders fit; at 1 only
|
|
/// one does; at 0 the guard is off with the hooks.
|
|
pub require_reply: bool,
|
|
/// Hook server allowlist. See [`HookServers`] for variant semantics.
|
|
/// Default (env unset/empty) is `None` — hooks are off unless the
|
|
/// operator explicitly opts in.
|
|
pub hook_servers: HookServers,
|
|
pub api_key: String,
|
|
pub model: String,
|
|
pub base_url: String,
|
|
pub anthropic_api_version: String,
|
|
/// OpenAI endpoint selection. See [`OpenAiApi`].
|
|
pub openai_api: OpenAiApi,
|
|
/// Prefer mesh-llm's virtual `mesh` model when the configured/effective
|
|
/// OpenAI model is `auto` and the live model catalog advertises it.
|
|
/// Set by Buzz's relay-mesh provider via
|
|
/// `BUZZ_AGENT_PREFER_MESH_FOR_AUTO=1`; other providers keep their
|
|
/// existing `auto` semantics.
|
|
pub prefer_mesh_for_auto: bool,
|
|
pub hints_enabled: bool,
|
|
/// Thinking/reasoning effort level. `None` = use provider default (no
|
|
/// thinking config sent). Set via `BUZZ_AGENT_THINKING_EFFORT`.
|
|
pub thinking_effort: Option<ThinkingEffort>,
|
|
/// Reasoning summary mode for the OpenAI Responses route. Controls the
|
|
/// `reasoning.summary` field emitted alongside `reasoning.effort`; only
|
|
/// takes effect when `thinking_effort` is also set. Default `Auto`.
|
|
/// Set via `BUZZ_AGENT_THINKING_SUMMARY`. Ignored on Anthropic, Chat
|
|
/// Completions, and OpenRouter routes.
|
|
pub thinking_summary: ThinkingSummary,
|
|
/// Emit Anthropic `cache_control` breakpoints on the stable prefix
|
|
/// (tools + system prompt) and the rolling conversation tail. Default on;
|
|
/// disable with `BUZZ_AGENT_PROMPT_CACHING=0`. Consulted on every route that
|
|
/// speaks the Anthropic caching dialect: first-party Anthropic, the
|
|
/// DatabricksV2 Claude route, and OpenRouter's `anthropic/*` models. The
|
|
/// Databricks gateway does not auto-cache, so without this the surfaced
|
|
/// `cache_read_input_tokens` is structurally always 0.
|
|
pub prompt_caching: bool,
|
|
}
|
|
|
|
impl Config {
|
|
pub fn from_env() -> Result<Self, String> {
|
|
let databricks_host = env("DATABRICKS_HOST");
|
|
let databricks_model = env("DATABRICKS_MODEL");
|
|
let provider = resolve_provider(
|
|
env("BUZZ_AGENT_PROVIDER").as_deref(),
|
|
env("ANTHROPIC_API_KEY").as_deref(),
|
|
env("OPENAI_COMPAT_API_KEY").as_deref(),
|
|
env("OPENROUTER_API_KEY").as_deref(),
|
|
)?;
|
|
|
|
// Universal model override — takes priority over provider-specific model
|
|
// env vars (ANTHROPIC_MODEL, OPENAI_COMPAT_MODEL, DATABRICKS_MODEL) when
|
|
// present. Set by the desktop from the persona/record to express explicit
|
|
// user intent; provider-specific vars serve as defaults for CLI/standalone use.
|
|
let buzz_agent_model = env("BUZZ_AGENT_MODEL");
|
|
|
|
// OPENAI_COMPAT_API is only read when provider=openai, so a stray
|
|
// bad value can't break an Anthropic-only deployment.
|
|
//
|
|
// Databricks borrows api_key as the *optional* `DATABRICKS_TOKEN` escape
|
|
// hatch — empty means "use OAuth PKCE." Legacy Databricks encodes the
|
|
// model in the URL path; Databricks v2 keeps it in the request body.
|
|
let (api_key, model, base_url, openai_api) = match provider {
|
|
Provider::Anthropic => (
|
|
req("ANTHROPIC_API_KEY")?,
|
|
resolve_model(
|
|
buzz_agent_model.as_deref(),
|
|
env("ANTHROPIC_MODEL").as_deref(),
|
|
)
|
|
.ok_or_else(|| "config: ANTHROPIC_MODEL required".to_string())?,
|
|
env_or("ANTHROPIC_BASE_URL", "https://api.anthropic.com"),
|
|
OpenAiApi::Auto, // unused for Anthropic
|
|
),
|
|
Provider::OpenAi => (
|
|
req("OPENAI_COMPAT_API_KEY")?,
|
|
resolve_model(
|
|
buzz_agent_model.as_deref(),
|
|
env("OPENAI_COMPAT_MODEL").as_deref(),
|
|
)
|
|
.ok_or_else(|| "config: OPENAI_COMPAT_MODEL required".to_string())?,
|
|
env_or("OPENAI_COMPAT_BASE_URL", "https://api.openai.com/v1"),
|
|
parse_openai_api(env("OPENAI_COMPAT_API").as_deref())?,
|
|
),
|
|
Provider::Databricks | Provider::DatabricksV2 => (
|
|
env("DATABRICKS_TOKEN").unwrap_or_default(),
|
|
resolve_model(buzz_agent_model.as_deref(), databricks_model.as_deref())
|
|
.ok_or_else(|| "config: DATABRICKS_MODEL required".to_string())?,
|
|
databricks_host.ok_or_else(|| "config: DATABRICKS_HOST required".to_string())?,
|
|
OpenAiApi::Chat, // only read by OpenAI/legacy Databricks dispatch
|
|
),
|
|
Provider::OpenRouter => (
|
|
req("OPENROUTER_API_KEY")?,
|
|
resolve_model(
|
|
buzz_agent_model.as_deref(),
|
|
env("OPENROUTER_MODEL").as_deref(),
|
|
)
|
|
.ok_or_else(|| "config: OPENROUTER_MODEL required".to_string())?,
|
|
env_or("OPENROUTER_BASE_URL", "https://openrouter.ai/api/v1"),
|
|
OpenAiApi::Chat, // OpenRouter uses Chat Completions only
|
|
),
|
|
};
|
|
let system_prompt = match (env("BUZZ_AGENT_SYSTEM_PROMPT"), env("BUZZ_AGENT_SYSTEM_PROMPT_FILE")) {
|
|
(Some(_), Some(_)) => return Err(
|
|
"config: BUZZ_AGENT_SYSTEM_PROMPT and BUZZ_AGENT_SYSTEM_PROMPT_FILE are mutually exclusive".into()),
|
|
(Some(s), _) => s,
|
|
(_, Some(p)) => std::fs::read_to_string(&p).map_err(|e| format!("config: read {p}: {e}"))?,
|
|
_ => DEFAULT_SYSTEM_PROMPT.to_owned(),
|
|
};
|
|
let cfg = Config {
|
|
provider,
|
|
system_prompt,
|
|
api_key,
|
|
model,
|
|
base_url,
|
|
anthropic_api_version: env_or("ANTHROPIC_API_VERSION", "2023-06-01"),
|
|
openai_api,
|
|
prefer_mesh_for_auto: parse_env("BUZZ_AGENT_PREFER_MESH_FOR_AUTO", 0u8)? != 0,
|
|
max_rounds: parse_env("BUZZ_AGENT_MAX_ROUNDS", 0)?,
|
|
max_output_tokens: parse_env("BUZZ_AGENT_MAX_OUTPUT_TOKENS", 32_768)?,
|
|
llm_timeout: Duration::from_secs(parse_env("BUZZ_AGENT_LLM_TIMEOUT_SECS", 240)?),
|
|
tool_timeout: Duration::from_secs(parse_env("BUZZ_AGENT_TOOL_TIMEOUT_SECS", 660)?),
|
|
mcp_init_timeout: Duration::from_secs(parse_env(
|
|
"BUZZ_AGENT_MCP_INIT_TIMEOUT_SECS",
|
|
30,
|
|
)?),
|
|
mcp_max_restart_attempts: parse_env("BUZZ_AGENT_MCP_RESTART_MAX_ATTEMPTS", 3u32)?,
|
|
mcp_restart_base_ms: parse_env("BUZZ_AGENT_MCP_RESTART_BASE_MS", 500u64)?,
|
|
mcp_restart_max_ms: parse_env("BUZZ_AGENT_MCP_RESTART_MAX_MS", 30_000u64)?,
|
|
max_sessions: parse_env("BUZZ_AGENT_MAX_SESSIONS", usize::MAX)?,
|
|
max_line_bytes: parse_env("BUZZ_AGENT_MAX_LINE_BYTES", 4 * 1024 * 1024)?,
|
|
max_history_bytes: parse_env("BUZZ_AGENT_MAX_HISTORY_BYTES", 16 * 1024 * 1024)?,
|
|
max_tool_result_text_bytes: parse_env(
|
|
"BUZZ_AGENT_MAX_TOOL_RESULT_TEXT_BYTES",
|
|
DEFAULT_TOOL_RESULT_TEXT_BYTES,
|
|
)?,
|
|
max_context_tokens: parse_env("BUZZ_AGENT_MAX_CONTEXT_TOKENS", 200_000u64)?,
|
|
max_handoffs: parse_env("BUZZ_AGENT_MAX_HANDOFFS", 10)?,
|
|
max_parallel_tools: parse_env("BUZZ_AGENT_MAX_PARALLEL_TOOLS", 8usize)?,
|
|
hook_timeout: Duration::from_millis(parse_env("BUZZ_AGENT_HOOK_TIMEOUT_MS", 2500u64)?),
|
|
stop_max_rejections: parse_env("BUZZ_AGENT_STOP_MAX_REJECTIONS", 3u32)?,
|
|
require_reply: parse_env("BUZZ_AGENT_REQUIRE_REPLY", 0u8)? != 0,
|
|
hook_servers: parse_hook_servers_env("MCP_HOOK_SERVERS"),
|
|
hints_enabled: parse_env("BUZZ_AGENT_NO_HINTS", 0u8)? == 0,
|
|
thinking_effort: parse_thinking_effort(env("BUZZ_AGENT_THINKING_EFFORT").as_deref())?,
|
|
thinking_summary: parse_thinking_summary(
|
|
env("BUZZ_AGENT_THINKING_SUMMARY").as_deref(),
|
|
)?,
|
|
prompt_caching: parse_env("BUZZ_AGENT_PROMPT_CACHING", 1u8)? != 0,
|
|
};
|
|
cfg.validate()?;
|
|
Ok(cfg)
|
|
}
|
|
|
|
/// Construct a minimal `Config` for model-catalog discovery.
|
|
///
|
|
/// Only the fields used by [`build_token_source`](crate::llm::build_token_source)
|
|
/// and the catalog HTTP helpers are meaningful; all others are set to
|
|
/// inert defaults. Never call `from_env` for discovery — it requires
|
|
/// `DATABRICKS_MODEL` and other fields that are irrelevant here.
|
|
pub fn for_discovery(provider: Provider, api_key: String, base_url: String) -> Self {
|
|
Self {
|
|
provider,
|
|
api_key,
|
|
base_url,
|
|
model: String::new(),
|
|
system_prompt: String::new(),
|
|
anthropic_api_version: "2023-06-01".into(),
|
|
openai_api: OpenAiApi::Chat,
|
|
prefer_mesh_for_auto: false,
|
|
max_rounds: 0,
|
|
max_output_tokens: 1,
|
|
llm_timeout: Duration::from_secs(30),
|
|
tool_timeout: Duration::from_secs(30),
|
|
mcp_init_timeout: Duration::from_secs(30),
|
|
mcp_max_restart_attempts: 0,
|
|
mcp_restart_base_ms: 0,
|
|
mcp_restart_max_ms: 0,
|
|
max_sessions: 1,
|
|
max_line_bytes: 4 * 1024 * 1024,
|
|
max_history_bytes: 16 * 1024 * 1024,
|
|
max_tool_result_text_bytes: 50 * 1024,
|
|
max_context_tokens: 200_001,
|
|
max_handoffs: 0,
|
|
max_parallel_tools: 1,
|
|
hook_timeout: Duration::from_secs(1),
|
|
stop_max_rejections: 0,
|
|
require_reply: false,
|
|
hook_servers: HookServers::None,
|
|
hints_enabled: false,
|
|
thinking_effort: None,
|
|
thinking_summary: ThinkingSummary::Auto,
|
|
prompt_caching: false,
|
|
}
|
|
}
|
|
|
|
fn validate(&self) -> Result<(), String> {
|
|
const MIN_HISTORY_BYTES: usize = 4096;
|
|
const MIN_LINE_BYTES: usize = 1024;
|
|
const MIN_TOOL_RESULT_TEXT_BYTES: usize = 1024;
|
|
const MIN_TIMEOUT: Duration = Duration::from_secs(1);
|
|
|
|
if self.max_output_tokens < 1 {
|
|
return Err("config: BUZZ_AGENT_MAX_OUTPUT_TOKENS must be >= 1".into());
|
|
}
|
|
if self.max_context_tokens <= u64::from(self.max_output_tokens) {
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_MAX_CONTEXT_TOKENS ({}) must be > BUZZ_AGENT_MAX_OUTPUT_TOKENS ({}) — the context window must leave room for the response",
|
|
self.max_context_tokens, self.max_output_tokens
|
|
));
|
|
}
|
|
if self.max_history_bytes < MIN_HISTORY_BYTES {
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_MAX_HISTORY_BYTES must be >= {MIN_HISTORY_BYTES}"
|
|
));
|
|
}
|
|
if self.max_history_bytes < MAX_PROMPT_BYTES {
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_MAX_HISTORY_BYTES ({}) must be >= MAX_PROMPT_BYTES ({MAX_PROMPT_BYTES})",
|
|
self.max_history_bytes
|
|
));
|
|
}
|
|
if self.max_line_bytes < MIN_LINE_BYTES {
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_MAX_LINE_BYTES must be >= {MIN_LINE_BYTES}"
|
|
));
|
|
}
|
|
if self.max_tool_result_text_bytes < MIN_TOOL_RESULT_TEXT_BYTES
|
|
|| self.max_tool_result_text_bytes > MAX_TOOL_RESULT_BYTES
|
|
{
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_MAX_TOOL_RESULT_TEXT_BYTES must be in {MIN_TOOL_RESULT_TEXT_BYTES}..={MAX_TOOL_RESULT_BYTES}"
|
|
));
|
|
}
|
|
if self.llm_timeout < MIN_TIMEOUT {
|
|
return Err("config: BUZZ_AGENT_LLM_TIMEOUT_SECS must be >= 1".into());
|
|
}
|
|
if self.tool_timeout < MIN_TIMEOUT {
|
|
return Err("config: BUZZ_AGENT_TOOL_TIMEOUT_SECS must be >= 1".into());
|
|
}
|
|
if self.mcp_init_timeout < MIN_TIMEOUT {
|
|
return Err("config: BUZZ_AGENT_MCP_INIT_TIMEOUT_SECS must be >= 1".into());
|
|
}
|
|
if self.max_parallel_tools < 1 {
|
|
return Err("config: BUZZ_AGENT_MAX_PARALLEL_TOOLS must be >= 1".into());
|
|
}
|
|
if self.mcp_max_restart_attempts < 1 {
|
|
return Err("config: BUZZ_AGENT_MCP_RESTART_MAX_ATTEMPTS must be >= 1".into());
|
|
}
|
|
if self.mcp_restart_base_ms < 1 {
|
|
return Err("config: BUZZ_AGENT_MCP_RESTART_BASE_MS must be >= 1".into());
|
|
}
|
|
if self.mcp_restart_max_ms < self.mcp_restart_base_ms {
|
|
return Err(
|
|
"config: BUZZ_AGENT_MCP_RESTART_MAX_MS must be >= BUZZ_AGENT_MCP_RESTART_BASE_MS"
|
|
.into(),
|
|
);
|
|
}
|
|
// Provider-level effort validation (fail-fast, clear error).
|
|
// `none`/`minimal` are not Anthropic values — rejected at startup.
|
|
//
|
|
// OpenAI, Databricks, and DatabricksV2 defer effort validation to request-time routing:
|
|
// availability is model-dependent, and `session/set_model` can change the effective model
|
|
// after startup. `normalize_effort_for_openai_route` / `normalize_effort_for_anthropic_route`
|
|
// apply route-aware normalization in `llm.rs` when building each request.
|
|
if let Some(effort) = self.thinking_effort {
|
|
let is_pure_anthropic = matches!(self.provider, Provider::Anthropic);
|
|
if is_pure_anthropic && matches!(effort, ThinkingEffort::None | ThinkingEffort::Minimal)
|
|
{
|
|
return Err(format!(
|
|
"config: BUZZ_AGENT_THINKING_EFFORT={} is not valid for Anthropic providers \
|
|
(allowed: low|medium|high|xhigh|max)",
|
|
effort.openai_effort_str()
|
|
));
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
}
|
|
|
|
fn env(k: &str) -> Option<String> {
|
|
std::env::var(k).ok()
|
|
}
|
|
|
|
fn env_or(k: &str, d: &str) -> String {
|
|
env(k).unwrap_or_else(|| d.into())
|
|
}
|
|
|
|
fn req(k: &str) -> Result<String, String> {
|
|
env(k).ok_or_else(|| format!("config: {k} required"))
|
|
}
|
|
|
|
/// Returns the first present value. `explicit_override` (BUZZ_AGENT_MODEL,
|
|
/// set by the desktop from the persona/record) wins over `provider_default`
|
|
/// (provider-specific env var that may be inherited from the shell).
|
|
/// Returns `None` when both are absent so the caller can supply a
|
|
/// provider-specific error message.
|
|
fn resolve_model(
|
|
explicit_override: Option<&str>,
|
|
provider_default: Option<&str>,
|
|
) -> Option<String> {
|
|
explicit_override.or(provider_default).map(str::to_owned)
|
|
}
|
|
|
|
fn present_nonempty(v: Option<&str>) -> bool {
|
|
v.map(str::trim).is_some_and(|s| !s.is_empty())
|
|
}
|
|
|
|
fn resolve_provider(
|
|
requested: Option<&str>,
|
|
anthropic_key: Option<&str>,
|
|
openai_key: Option<&str>,
|
|
openrouter_key: Option<&str>,
|
|
) -> Result<Provider, String> {
|
|
match requested.map(str::trim).filter(|s| !s.is_empty()) {
|
|
Some(raw) => {
|
|
let normalized = raw.to_ascii_lowercase();
|
|
match normalized.as_str() {
|
|
"anthropic" if present_nonempty(anthropic_key) => Ok(Provider::Anthropic),
|
|
"anthropic" => Err(
|
|
"config: ANTHROPIC_API_KEY required".into(),
|
|
),
|
|
"openai" | "openai-compat" if present_nonempty(openai_key) => Ok(Provider::OpenAi),
|
|
"openai" | "openai-compat" => Err(
|
|
"config: OPENAI_COMPAT_API_KEY required".into(),
|
|
),
|
|
"databricks" => Ok(Provider::Databricks),
|
|
"databricks_v2" | "databricks-v2" => Ok(Provider::DatabricksV2),
|
|
"openrouter" if present_nonempty(openrouter_key) => Ok(Provider::OpenRouter),
|
|
"openrouter" => Err("config: OPENROUTER_API_KEY required".into()),
|
|
_ => Err(format!(
|
|
"config: BUZZ_AGENT_PROVIDER={raw} not supported"
|
|
)),
|
|
}
|
|
}
|
|
None => Err(
|
|
"config: BUZZ_AGENT_PROVIDER is required — set it to your provider (e.g. anthropic, openai, databricks)".into(),
|
|
),
|
|
}
|
|
}
|
|
|
|
/// Parse `OPENAI_COMPAT_API`. Pure (env-free) for testability; the
|
|
/// caller hands in the raw value.
|
|
fn parse_openai_api(raw: Option<&str>) -> Result<OpenAiApi, String> {
|
|
match raw.unwrap_or("auto").trim().to_ascii_lowercase().as_str() {
|
|
"chat" | "chat-completions" | "chat_completions" => Ok(OpenAiApi::Chat),
|
|
"responses" => Ok(OpenAiApi::Responses),
|
|
"auto" | "" => Ok(OpenAiApi::Auto),
|
|
other => Err(format!(
|
|
"config: OPENAI_COMPAT_API={other} not supported (use auto|chat|responses)"
|
|
)),
|
|
}
|
|
}
|
|
|
|
/// `true` when `base_url` is an official OpenAI host. Hosts on
|
|
/// `*.openai.com` get Responses under `Auto`; everything else (vLLM,
|
|
/// Ollama, OpenRouter, Block Gateway, …) gets Chat Completions.
|
|
/// Lookalike-safe: `api.openai.com.evil.example` returns `false`.
|
|
pub fn is_openai_host(base_url: &str) -> bool {
|
|
let rest = match base_url
|
|
.strip_prefix("https://")
|
|
.or_else(|| base_url.strip_prefix("http://"))
|
|
{
|
|
Some(r) => r,
|
|
None => return false,
|
|
};
|
|
let host = &rest[..rest.find(['/', ':']).unwrap_or(rest.len())];
|
|
host == "api.openai.com" || host.ends_with(".openai.com")
|
|
}
|
|
|
|
fn parse_env<T: std::str::FromStr>(key: &str, default: T) -> Result<T, String>
|
|
where
|
|
T::Err: std::fmt::Display,
|
|
{
|
|
env(key)
|
|
.map(|v| v.parse().map_err(|e| format!("config: {key}: {e}")))
|
|
.unwrap_or(Ok(default))
|
|
}
|
|
|
|
/// Hook-server allowlist parsed from a comma-separated env var.
|
|
/// - unset / empty / whitespace-only → `None` (no hooks enabled)
|
|
/// - `*` → `All` (every server eligible)
|
|
/// - `a,b,c` → `Only(["a","b","c"])`
|
|
#[derive(Debug, Clone)]
|
|
pub enum HookServers {
|
|
None,
|
|
All,
|
|
Only(Vec<String>),
|
|
}
|
|
|
|
impl HookServers {
|
|
/// Returns true iff `name` may receive hook calls.
|
|
pub fn allows(&self, name: &str) -> bool {
|
|
match self {
|
|
HookServers::None => false,
|
|
HookServers::All => true,
|
|
HookServers::Only(v) => v.iter().any(|s| s == name),
|
|
}
|
|
}
|
|
|
|
/// True if no hooks should ever fire — used to short-circuit dispatch.
|
|
pub fn is_disabled(&self) -> bool {
|
|
matches!(self, HookServers::None)
|
|
}
|
|
}
|
|
|
|
fn parse_hook_servers_env(key: &str) -> HookServers {
|
|
parse_hook_servers(env(key).as_deref())
|
|
}
|
|
|
|
/// Pure parser exposed for unit tests. `None` (env unset) and `Some("")`
|
|
/// (env set but empty) both yield `HookServers::None`.
|
|
fn parse_hook_servers(raw: Option<&str>) -> HookServers {
|
|
let raw = match raw {
|
|
Some(v) => v,
|
|
None => return HookServers::None,
|
|
};
|
|
let names: Vec<String> = raw
|
|
.split(',')
|
|
.map(|s| s.trim().to_owned())
|
|
.filter(|s| !s.is_empty())
|
|
.collect();
|
|
if names.is_empty() {
|
|
return HookServers::None;
|
|
}
|
|
// `*` is the wildcard — only honored when it's the sole entry. A mixed
|
|
// value like "*,foo" falls through to `Only(["*","foo"])`; "*" is not a
|
|
// legal MCP server name (it can't pass `valid_name`), so it never matches
|
|
// an actual server. This avoids silently widening scope on typos.
|
|
if names.len() == 1 && names[0] == "*" {
|
|
return HookServers::All;
|
|
}
|
|
HookServers::Only(names)
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn hook_servers_unset_is_none() {
|
|
assert!(matches!(parse_hook_servers(None), HookServers::None));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_empty_string_is_none() {
|
|
assert!(matches!(parse_hook_servers(Some("")), HookServers::None));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_whitespace_only_is_none() {
|
|
assert!(matches!(
|
|
parse_hook_servers(Some(" ,, ,")),
|
|
HookServers::None
|
|
));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_star_is_all() {
|
|
assert!(matches!(parse_hook_servers(Some("*")), HookServers::All));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_star_with_whitespace_is_all() {
|
|
assert!(matches!(
|
|
parse_hook_servers(Some(" * ")),
|
|
HookServers::All
|
|
));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_named_list() {
|
|
match parse_hook_servers(Some("foo,bar")) {
|
|
HookServers::Only(v) => assert_eq!(v, vec!["foo".to_owned(), "bar".to_owned()]),
|
|
other => panic!("expected Only, got {other:?}"),
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_trims_entries() {
|
|
match parse_hook_servers(Some(" foo , bar , ")) {
|
|
HookServers::Only(v) => assert_eq!(v, vec!["foo".to_owned(), "bar".to_owned()]),
|
|
other => panic!("expected Only, got {other:?}"),
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_star_mixed_is_literal() {
|
|
// `*,foo` is NOT a wildcard — it's a literal Only(["*","foo"]).
|
|
// No real server can be named `*`, so this never matches anything.
|
|
match parse_hook_servers(Some("*,foo")) {
|
|
HookServers::Only(v) => assert_eq!(v, vec!["*".to_owned(), "foo".to_owned()]),
|
|
other => panic!("expected Only, got {other:?}"),
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_allows_matches_named_only() {
|
|
let hs = parse_hook_servers(Some("foo,bar"));
|
|
assert!(hs.allows("foo"));
|
|
assert!(hs.allows("bar"));
|
|
assert!(!hs.allows("baz"));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_allows_matches_all() {
|
|
assert!(parse_hook_servers(Some("*")).allows("anything"));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_allows_blocks_when_none() {
|
|
assert!(!parse_hook_servers(None).allows("foo"));
|
|
}
|
|
|
|
#[test]
|
|
fn hook_servers_star_mixed_does_not_match_real_server() {
|
|
let hs = parse_hook_servers(Some("*,foo"));
|
|
// The literal "*" entry exists in Only, but no real server can
|
|
// be named "*" (rejected by the MCP server name validator).
|
|
assert!(hs.allows("foo"));
|
|
assert!(!hs.allows("bar"));
|
|
// Allowed strictly only as a literal match — defense-in-depth
|
|
// expectation for callers.
|
|
assert!(hs.allows("*"));
|
|
}
|
|
|
|
#[test]
|
|
fn parse_openai_api_values() {
|
|
use OpenAiApi::*;
|
|
for (raw, want) in [
|
|
(None, Ok(Auto)),
|
|
(Some("auto"), Ok(Auto)),
|
|
(Some(" AUTO "), Ok(Auto)),
|
|
(Some(""), Ok(Auto)),
|
|
(Some("chat"), Ok(Chat)),
|
|
(Some("chat-completions"), Ok(Chat)),
|
|
(Some("Responses"), Ok(Responses)),
|
|
] {
|
|
assert_eq!(parse_openai_api(raw), want, "raw={raw:?}");
|
|
}
|
|
let err = parse_openai_api(Some("nope")).unwrap_err();
|
|
assert!(err.contains("OPENAI_COMPAT_API=nope"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_keeps_requested_provider_when_token_present() {
|
|
assert_eq!(
|
|
resolve_provider(Some("anthropic"), Some("sk-ant"), None, None).unwrap(),
|
|
Provider::Anthropic
|
|
);
|
|
assert_eq!(
|
|
resolve_provider(Some("openai"), None, Some("sk-openai"), None).unwrap(),
|
|
Provider::OpenAi
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_errors_when_requested_provider_key_missing() {
|
|
// No fallback — missing key returns an error regardless of Databricks availability.
|
|
let err = resolve_provider(Some("anthropic"), None, None, None).unwrap_err();
|
|
assert!(err.contains("ANTHROPIC_API_KEY required"), "{err}");
|
|
|
|
let err = resolve_provider(Some("openai-compat"), None, Some(" "), None).unwrap_err();
|
|
assert!(err.contains("OPENAI_COMPAT_API_KEY required"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_errors_when_provider_env_absent() {
|
|
// No implicit inference — absent BUZZ_AGENT_PROVIDER is an error.
|
|
let err = resolve_provider(None, None, None, None).unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_PROVIDER is required"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_requires_databricks_host_and_model_for_fallback() {
|
|
// Renamed: verify the explicit databricks provider path works correctly.
|
|
// When BUZZ_AGENT_PROVIDER=databricks, resolve_provider succeeds regardless
|
|
// of DATABRICKS_HOST/MODEL (those are validated later in from_env()).
|
|
assert_eq!(
|
|
resolve_provider(Some("databricks"), None, None, None).unwrap(),
|
|
Provider::Databricks
|
|
);
|
|
// Missing key for other providers still errors — no Databricks fallback.
|
|
let err = resolve_provider(Some("openai"), None, None, None).unwrap_err();
|
|
assert!(err.contains("OPENAI_COMPAT_API_KEY required"), "{err}");
|
|
let err = resolve_provider(None, None, None, None).unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_PROVIDER is required"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_unsupported_error_preserves_user_casing() {
|
|
let err = resolve_provider(Some("OpenAIish"), None, None, None).unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_PROVIDER=OpenAIish"));
|
|
}
|
|
|
|
#[test]
|
|
fn is_openai_host_matrix() {
|
|
// Lookalike-safe: `api.openai.com.evil.example` and malformed URLs
|
|
// are treated as non-OpenAI (which falls back to Chat Completions).
|
|
for (url, want) in [
|
|
("https://api.openai.com/v1", true),
|
|
("https://api.openai.com", true),
|
|
("http://eu.api.openai.com/v1", true),
|
|
("http://localhost:11434/v1", false),
|
|
("https://openrouter.ai/api/v1", false),
|
|
("https://gateway.block.example/v1", false),
|
|
("https://api.openai.com.evil.example/v1", false),
|
|
("not a url", false),
|
|
] {
|
|
assert_eq!(is_openai_host(url), want, "url={url}");
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_model_prefers_explicit_override() {
|
|
let result = resolve_model(Some("override-model"), Some("provider-model"));
|
|
assert_eq!(result.as_deref(), Some("override-model"));
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_model_falls_back_to_provider_default() {
|
|
let result = resolve_model(None, Some("provider-model"));
|
|
assert_eq!(result.as_deref(), Some("provider-model"));
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_model_returns_none_when_both_absent() {
|
|
let result = resolve_model(None, None);
|
|
assert!(result.is_none());
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_effort_round_trips_all_values() {
|
|
for (raw, expected) in [
|
|
("none", ThinkingEffort::None),
|
|
("minimal", ThinkingEffort::Minimal),
|
|
("low", ThinkingEffort::Low),
|
|
("medium", ThinkingEffort::Medium),
|
|
("high", ThinkingEffort::High),
|
|
("xhigh", ThinkingEffort::XHigh),
|
|
("max", ThinkingEffort::Max),
|
|
] {
|
|
assert_eq!(
|
|
parse_thinking_effort(Some(raw)).unwrap(),
|
|
Some(expected),
|
|
"raw={raw:?}"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_effort_none_and_empty_yield_none() {
|
|
assert_eq!(parse_thinking_effort(None).unwrap(), None);
|
|
assert_eq!(parse_thinking_effort(Some("")).unwrap(), None);
|
|
assert_eq!(parse_thinking_effort(Some(" ")).unwrap(), None);
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_effort_is_case_insensitive() {
|
|
assert_eq!(
|
|
parse_thinking_effort(Some("HIGH")).unwrap(),
|
|
Some(ThinkingEffort::High)
|
|
);
|
|
assert_eq!(
|
|
parse_thinking_effort(Some(" Medium ")).unwrap(),
|
|
Some(ThinkingEffort::Medium)
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_effort_rejects_unknown_value() {
|
|
let err = parse_thinking_effort(Some("extreme")).unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_THINKING_EFFORT=extreme"), "{err}");
|
|
assert!(
|
|
err.contains("none|minimal|low|medium|high|xhigh|max"),
|
|
"{err}"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_summary_round_trips_all_values() {
|
|
for (raw, expected) in [
|
|
("auto", ThinkingSummary::Auto),
|
|
("concise", ThinkingSummary::Concise),
|
|
("detailed", ThinkingSummary::Detailed),
|
|
] {
|
|
assert_eq!(
|
|
parse_thinking_summary(Some(raw)).unwrap(),
|
|
expected,
|
|
"raw={raw:?}"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_summary_unset_and_empty_yield_auto() {
|
|
assert_eq!(parse_thinking_summary(None).unwrap(), ThinkingSummary::Auto);
|
|
assert_eq!(
|
|
parse_thinking_summary(Some("")).unwrap(),
|
|
ThinkingSummary::Auto
|
|
);
|
|
assert_eq!(
|
|
parse_thinking_summary(Some(" ")).unwrap(),
|
|
ThinkingSummary::Auto
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_summary_is_case_insensitive() {
|
|
assert_eq!(
|
|
parse_thinking_summary(Some("DETAILED")).unwrap(),
|
|
ThinkingSummary::Detailed
|
|
);
|
|
assert_eq!(
|
|
parse_thinking_summary(Some(" Concise ")).unwrap(),
|
|
ThinkingSummary::Concise
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn parse_thinking_summary_rejects_unknown_value() {
|
|
let err = parse_thinking_summary(Some("verbose")).unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_THINKING_SUMMARY=verbose"), "{err}");
|
|
assert!(err.contains("auto|concise|detailed"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn thinking_summary_as_str_mapping() {
|
|
assert_eq!(ThinkingSummary::Auto.as_str(), "auto");
|
|
assert_eq!(ThinkingSummary::Concise.as_str(), "concise");
|
|
assert_eq!(ThinkingSummary::Detailed.as_str(), "detailed");
|
|
}
|
|
|
|
#[test]
|
|
fn thinking_effort_anthropic_budget_tokens_mapping() {
|
|
assert_eq!(ThinkingEffort::Low.anthropic_budget_tokens(), 1_024);
|
|
assert_eq!(ThinkingEffort::Medium.anthropic_budget_tokens(), 8_192);
|
|
assert_eq!(ThinkingEffort::High.anthropic_budget_tokens(), 32_768);
|
|
// XHigh and Max clamp to the high budget value for manual-budget models.
|
|
assert_eq!(ThinkingEffort::XHigh.anthropic_budget_tokens(), 32_768);
|
|
assert_eq!(ThinkingEffort::Max.anthropic_budget_tokens(), 32_768);
|
|
// None/Minimal are rejected at startup for Anthropic; defensive zero.
|
|
assert_eq!(ThinkingEffort::None.anthropic_budget_tokens(), 0);
|
|
assert_eq!(ThinkingEffort::Minimal.anthropic_budget_tokens(), 0);
|
|
}
|
|
|
|
#[test]
|
|
fn thinking_effort_openai_effort_str_mapping() {
|
|
assert_eq!(ThinkingEffort::None.openai_effort_str(), "none");
|
|
assert_eq!(ThinkingEffort::Minimal.openai_effort_str(), "minimal");
|
|
assert_eq!(ThinkingEffort::Low.openai_effort_str(), "low");
|
|
assert_eq!(ThinkingEffort::Medium.openai_effort_str(), "medium");
|
|
assert_eq!(ThinkingEffort::High.openai_effort_str(), "high");
|
|
assert_eq!(ThinkingEffort::XHigh.openai_effort_str(), "xhigh");
|
|
assert_eq!(ThinkingEffort::Max.openai_effort_str(), "max");
|
|
}
|
|
|
|
#[test]
|
|
fn thinking_effort_anthropic_effort_str_mapping() {
|
|
assert_eq!(ThinkingEffort::Low.anthropic_effort_str(), "low");
|
|
assert_eq!(ThinkingEffort::Medium.anthropic_effort_str(), "medium");
|
|
assert_eq!(ThinkingEffort::High.anthropic_effort_str(), "high");
|
|
assert_eq!(ThinkingEffort::XHigh.anthropic_effort_str(), "xhigh");
|
|
assert_eq!(ThinkingEffort::Max.anthropic_effort_str(), "max");
|
|
// Defensive fallback for invalid Anthropic values (caught at startup validation).
|
|
assert_eq!(ThinkingEffort::None.anthropic_effort_str(), "low");
|
|
assert_eq!(ThinkingEffort::Minimal.anthropic_effort_str(), "low");
|
|
}
|
|
|
|
#[test]
|
|
fn thinking_effort_ord_ordering() {
|
|
// PartialOrd/Ord must reflect the ordered hierarchy.
|
|
assert!(ThinkingEffort::None < ThinkingEffort::Minimal);
|
|
assert!(ThinkingEffort::Minimal < ThinkingEffort::Low);
|
|
assert!(ThinkingEffort::Low < ThinkingEffort::Medium);
|
|
assert!(ThinkingEffort::Medium < ThinkingEffort::High);
|
|
assert!(ThinkingEffort::High < ThinkingEffort::XHigh);
|
|
assert!(ThinkingEffort::XHigh < ThinkingEffort::Max);
|
|
}
|
|
|
|
// ---- anthropic_thinking_config helper — per-family tests ----
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_claude3_emits_budget_tokens() {
|
|
// Claude 3.x → `thinking.budget_tokens`; clamped to min(level_budget, max_output - 1024).
|
|
// max_output_tokens = 4096: headroom = 4096 - 1024 = 3072; High budget (32768) → 3072.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-3-7-sonnet-20250219", ThinkingEffort::High, 4096);
|
|
let t = thinking.expect("thinking field must be present for claude-3");
|
|
assert_eq!(t["type"], "enabled");
|
|
assert_eq!(t["budget_tokens"], 3072); // capped: min(32768, 4096-1024)
|
|
assert!(
|
|
output_config.is_none(),
|
|
"output_config must be absent for claude-3"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_claude3_omits_thinking_when_max_output_too_small() {
|
|
// max_output_tokens = 2047: headroom = 2047 - 1024 = 1023 < 1024 → omit thinking.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-3-7-sonnet-20250219", ThinkingEffort::High, 2047);
|
|
assert!(
|
|
thinking.is_none(),
|
|
"thinking must be omitted when max_output_tokens - 1024 < 1024 (budget would starve answer)"
|
|
);
|
|
assert!(output_config.is_none());
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_claude3_emits_thinking_at_boundary_2048() {
|
|
// max_output_tokens = 2048: headroom = 2048 - 1024 = 1024 ≥ 1024 → emit budget = 1024.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-3-7-sonnet-20250219", ThinkingEffort::High, 2048);
|
|
let t = thinking.expect("thinking must be present when max_output_tokens = 2048");
|
|
assert_eq!(t["budget_tokens"], 1024); // min(32768, 2048-1024) = 1024
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_claude3_budget_uncapped_when_fits() {
|
|
// High budget fits comfortably under a large max_output_tokens.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-3-7-sonnet-20250219", ThinkingEffort::High, 65_536);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["budget_tokens"], 32_768);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_8_emits_adaptive_and_effort() {
|
|
// Opus 4.8 — adaptive family. Requires thinking:{type:"adaptive"} to enable thinking.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-8", ThinkingEffort::High, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-opus-4-8");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-opus-4-8");
|
|
assert_eq!(oc["effort"], "high");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_7_emits_adaptive_and_effort() {
|
|
// Opus 4.7 — adaptive family.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-7", ThinkingEffort::Medium, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-opus-4-7");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-opus-4-7");
|
|
assert_eq!(oc["effort"], "medium");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_sonnet_5_emits_adaptive_and_effort() {
|
|
// Sonnet 5 — adaptive family.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-sonnet-5-20250901", ThinkingEffort::Low, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-sonnet-5");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-sonnet-5");
|
|
assert_eq!(oc["effort"], "low");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_sonnet_4_6_emits_adaptive_and_effort() {
|
|
// Sonnet 4.6 — adaptive family. Docs explicitly list "Combine effort with adaptive thinking."
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-sonnet-4-6", ThinkingEffort::High, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-sonnet-4-6");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-sonnet-4-6");
|
|
assert_eq!(oc["effort"], "high");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_5_emits_manual_budget() {
|
|
// Opus 4.5 — manual budget (NOT adaptive; effort page: "uses manual thinking").
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-5", ThinkingEffort::High, 65_536);
|
|
let t = thinking.expect("thinking must be present for claude-opus-4-5");
|
|
assert_eq!(t["type"], "enabled");
|
|
assert_eq!(t["budget_tokens"], 32_768); // High budget fits under 65536
|
|
assert!(
|
|
output_config.is_none(),
|
|
"output_config must be absent for claude-opus-4-5 (manual budget)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_5_budget_capped() {
|
|
// Opus 4.5 manual budget is clamped to min(level_budget, max_output_tokens - 1024).
|
|
// max_output_tokens = 4096: headroom = 4096 - 1024 = 3072; High budget (32768) → 3072.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-opus-4-5", ThinkingEffort::High, 4096);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["budget_tokens"], 3072); // min(32768, 4096-1024)
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_5_omits_thinking_when_max_output_1025() {
|
|
// max_output_tokens = 1025: headroom = 1025 - 1024 = 1 < 1024 → omit thinking.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-opus-4-5", ThinkingEffort::High, 1025);
|
|
assert!(
|
|
thinking.is_none(),
|
|
"thinking must be omitted when max_output_tokens - 1024 < 1024"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_manual_budget_low_emits_1024_when_fits() {
|
|
// Low budget (1024 tokens) exactly fits when max_output_tokens = 2048.
|
|
// headroom = 2048 - 1024 = 1024; min(1024, 1024) = 1024 ≥ 1024 → emit.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-3-7-sonnet-20250219", ThinkingEffort::Low, 2048);
|
|
let t = thinking.expect("Low budget (1024) must be emitted when max_output_tokens = 2048");
|
|
assert_eq!(t["budget_tokens"], 1024);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_unknown_claude_omits_both_fields() {
|
|
// An unknown/future "claude-*" name that is not in the allowlist → omit both fields.
|
|
// This prevents sending an unverified shape to an unrecognized model.
|
|
// Includes Opus 4.9 (future version), which is NOT in the doc-verified adaptive list.
|
|
for model in &[
|
|
"claude-haiku-4-5",
|
|
"claude-sonnet-4-5",
|
|
"claude-unknown-9-1",
|
|
"claude-future-model",
|
|
"claude-opus-4-9",
|
|
] {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config(model, ThinkingEffort::High, 32_768);
|
|
assert!(
|
|
thinking.is_none(),
|
|
"thinking must be absent for unverified claude model: {model}"
|
|
);
|
|
assert!(
|
|
output_config.is_none(),
|
|
"output_config must be absent for unverified claude model: {model}"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_non_claude_omits_both_fields() {
|
|
// Non-Anthropic model names (gpt-5, llama, etc.) → omit both fields.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("gpt-4o-mini", ThinkingEffort::High, 32_768);
|
|
assert!(
|
|
thinking.is_none(),
|
|
"thinking must be absent for non-claude model"
|
|
);
|
|
assert!(
|
|
output_config.is_none(),
|
|
"output_config must be absent for non-claude model"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_databricks_prefix_stripped_for_claude3() {
|
|
// Databricks gateway prefixes like "databricks-claude-3-..." must be stripped.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("databricks-claude-3-5-sonnet", ThinkingEffort::Low, 8_192);
|
|
let t = thinking.expect("thinking must be present after stripping databricks- prefix");
|
|
assert_eq!(t["type"], "enabled");
|
|
assert!(output_config.is_none());
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_databricks_prefix_stripped_for_opus_4_7() {
|
|
// Databricks gateway prefix stripping applies to adaptive Claude families too.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("databricks-claude-opus-4-7", ThinkingEffort::High, 32_768);
|
|
let t = thinking
|
|
.expect("thinking:{type:adaptive} must be present for databricks-claude-opus-4-7");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc =
|
|
output_config.expect("output_config must be present for databricks-claude-opus-4-7");
|
|
assert_eq!(oc["effort"], "high");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_databricks_prefix_stripped_for_opus_4_8() {
|
|
// Databricks gateway prefix stripping applies to Opus 4.8 too.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("databricks-claude-opus-4-8", ThinkingEffort::Medium, 32_768);
|
|
let t = thinking
|
|
.expect("thinking:{type:adaptive} must be present for databricks-claude-opus-4-8");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc =
|
|
output_config.expect("output_config must be present for databricks-claude-opus-4-8");
|
|
assert_eq!(oc["effort"], "medium");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_goose_prefix_stripped_for_fable_5() {
|
|
// "goose-" catalog prefix must be stripped so goose-claude-fable-5 routes to
|
|
// the adaptive + xhigh/max bucket, not the "unknown model → (None, None)" path.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("goose-claude-fable-5", ThinkingEffort::Max, 32_768);
|
|
let t =
|
|
thinking.expect("thinking:{type:adaptive} must be present for goose-claude-fable-5");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for goose-claude-fable-5");
|
|
assert_eq!(oc["effort"], "max");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_goose_prefix_stripped_for_sonnet_5() {
|
|
// Adaptive xhigh model via goose- prefix.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("goose-claude-sonnet-5", ThinkingEffort::XHigh, 32_768);
|
|
let t =
|
|
thinking.expect("thinking:{type:adaptive} must be present for goose-claude-sonnet-5");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for goose-claude-sonnet-5");
|
|
assert_eq!(oc["effort"], "xhigh");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_arbitrary_prefix_stripped_for_opus_4_7() {
|
|
// team-x-claude-opus-4-7: first claude- token at index 7 → strips "team-x-"
|
|
// Verifies the arbitrary-prefix normalization reaches anthropic_thinking_config
|
|
// end-to-end: UI exposes max as valid, and runtime must honor it.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("team-x-claude-opus-4-7", ThinkingEffort::Max, 32_768);
|
|
let t =
|
|
thinking.expect("thinking:{type:adaptive} must be present for team-x-claude-opus-4-7");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for team-x-claude-opus-4-7");
|
|
assert_eq!(oc["effort"], "max");
|
|
}
|
|
|
|
// ---- anthropic_thinking_config: display:"summarized" in all enabled shapes ----
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_adaptive_emits_display_summarized() {
|
|
// Adaptive families (Opus 4.7, Sonnet 5, Fable 5, etc.) must include
|
|
// display:"summarized" so thinking text is returned, not omitted.
|
|
for model in &[
|
|
"claude-opus-4-7",
|
|
"claude-opus-4-8",
|
|
"claude-sonnet-5-20250901",
|
|
"claude-fable-5",
|
|
"claude-mythos-5",
|
|
] {
|
|
let (thinking, _) = anthropic_thinking_config(model, ThinkingEffort::High, 32_768);
|
|
let t = thinking
|
|
.unwrap_or_else(|| panic!("thinking must be present for adaptive model {model}"));
|
|
assert_eq!(
|
|
t["display"], "summarized",
|
|
"display:summarized must be present for adaptive model {model}: got {t}"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_manual_budget_emits_display_summarized() {
|
|
// Manual-budget families (claude-3.x, opus-4-5) must also include
|
|
// display:"summarized" so thinking text is returned.
|
|
for model in &["claude-3-7-sonnet-20250219", "claude-opus-4-5"] {
|
|
let (thinking, _) = anthropic_thinking_config(model, ThinkingEffort::High, 65_536);
|
|
let t = thinking.unwrap_or_else(|| {
|
|
panic!("thinking must be present for manual-budget model {model}")
|
|
});
|
|
assert_eq!(
|
|
t["display"], "summarized",
|
|
"display:summarized must be present for manual-budget model {model}: got {t}"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_omitted_when_no_thinking_has_no_display_field() {
|
|
// Models that don't produce a thinking field at all should have no display key.
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-haiku-4-5", ThinkingEffort::High, 32_768);
|
|
assert!(
|
|
thinking.is_none(),
|
|
"thinking must be absent for unknown model"
|
|
);
|
|
}
|
|
|
|
// ---- clamp_adaptive_effort — per-model clamping tests ----
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_passes_through_for_opus_4_7() {
|
|
// Opus 4.7 supports xhigh — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-7", ThinkingEffort::XHigh),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_passes_through_for_opus_4_8() {
|
|
// Opus 4.8 supports xhigh — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-8", ThinkingEffort::XHigh),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_passes_through_for_sonnet_5() {
|
|
// Sonnet 5 supports xhigh — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-sonnet-5-20250901", ThinkingEffort::XHigh),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_clamped_to_high_for_opus_4_6() {
|
|
// Opus 4.6 does NOT support xhigh (only low/medium/high/max) — clamp to high.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-6", ThinkingEffort::XHigh),
|
|
ThinkingEffort::High
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_clamped_to_high_for_sonnet_4_6() {
|
|
// Sonnet 4.6 does NOT support xhigh — clamp to high.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-sonnet-4-6", ThinkingEffort::XHigh),
|
|
ThinkingEffort::High
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_opus_4_6() {
|
|
// Opus 4.6 supports max — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-6", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_opus_4_7() {
|
|
// Opus 4.7 supports max — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-7", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_opus_4_8() {
|
|
// Opus 4.8 supports max — no clamping.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-opus-4-8", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_low_medium_high_never_clamped() {
|
|
// low/medium/high pass through for all adaptive models.
|
|
for model in &[
|
|
"claude-opus-4-6",
|
|
"claude-opus-4-7",
|
|
"claude-opus-4-8",
|
|
"claude-sonnet-5-20250901",
|
|
"claude-sonnet-4-6",
|
|
] {
|
|
for effort in [
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
] {
|
|
assert_eq!(
|
|
clamp_adaptive_effort(model, effort),
|
|
effort,
|
|
"model={model} effort={effort:?}"
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
// ---- anthropic_thinking_config — xhigh/max body-shape assertions ----
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_8_xhigh_emits_xhigh_effort() {
|
|
// Opus 4.8 supports xhigh; output_config.effort must be "xhigh".
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-8", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-opus-4-8");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-opus-4-8");
|
|
assert_eq!(oc["effort"], "xhigh");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_8_max_emits_max_effort() {
|
|
// Opus 4.8 supports max; output_config.effort must be "max".
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-8", ThinkingEffort::Max, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-opus-4-8");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-opus-4-8");
|
|
assert_eq!(oc["effort"], "max");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_7_xhigh_emits_xhigh_effort() {
|
|
// Opus 4.7 supports xhigh.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-7", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.unwrap();
|
|
assert_eq!(oc["effort"], "xhigh");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_6_xhigh_clamps_to_high() {
|
|
// Opus 4.6 does NOT support xhigh → clamp to high.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-6", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.unwrap();
|
|
assert_eq!(
|
|
oc["effort"], "high",
|
|
"xhigh must clamp to high for claude-opus-4-6"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_opus_4_6_max_passes_through() {
|
|
// Opus 4.6 supports max — passes through without clamping.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-opus-4-6", ThinkingEffort::Max, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.unwrap();
|
|
assert_eq!(oc["effort"], "max");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_manual_bucket_xhigh_clamps_to_high_budget() {
|
|
// Manual-budget models (claude-3*, opus-4-5): xhigh clamps to high budget (32_768).
|
|
for model in &["claude-3-7-sonnet-20250219", "claude-opus-4-5"] {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config(model, ThinkingEffort::XHigh, 65_536);
|
|
let t = thinking.expect("thinking must be present");
|
|
assert_eq!(t["type"], "enabled");
|
|
assert_eq!(
|
|
t["budget_tokens"], 32_768,
|
|
"xhigh must clamp to high budget for manual model {model}"
|
|
);
|
|
assert!(output_config.is_none());
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_manual_bucket_max_clamps_to_high_budget() {
|
|
// Manual-budget models: max also clamps to high budget (32_768).
|
|
let (thinking, _) =
|
|
anthropic_thinking_config("claude-opus-4-5", ThinkingEffort::Max, 65_536);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "enabled");
|
|
assert_eq!(t["budget_tokens"], 32_768);
|
|
}
|
|
|
|
// ---- provider-level validation tests ----
|
|
|
|
/// Build a minimal Config with the given provider and thinking_effort, bypassing from_env().
|
|
/// Uses `Config::for_discovery` as a base and patches the fields we care about.
|
|
fn make_config_for_validation(
|
|
provider: Provider,
|
|
thinking_effort: Option<ThinkingEffort>,
|
|
) -> Config {
|
|
let mut cfg = Config::for_discovery(provider, "key".into(), "https://example.com".into());
|
|
cfg.model = "some-model".into();
|
|
cfg.thinking_effort = thinking_effort;
|
|
// for_discovery sets max_output_tokens=1 and max_context_tokens=200_001 which satisfies
|
|
// the context > output constraint. Adjust to something valid for further checks.
|
|
cfg.max_output_tokens = 1024;
|
|
cfg.max_context_tokens = 200_000 + 1024;
|
|
// Restore mandatory positive values that for_discovery zeroes out.
|
|
cfg.mcp_max_restart_attempts = 1;
|
|
cfg.mcp_restart_base_ms = 1;
|
|
cfg.mcp_restart_max_ms = 1;
|
|
cfg.max_parallel_tools = 1;
|
|
cfg.llm_timeout = Duration::from_secs(1);
|
|
cfg.tool_timeout = Duration::from_secs(1);
|
|
cfg.mcp_init_timeout = Duration::from_secs(1);
|
|
cfg
|
|
}
|
|
|
|
#[test]
|
|
fn validate_rejects_none_effort_for_anthropic() {
|
|
let cfg = make_config_for_validation(Provider::Anthropic, Some(ThinkingEffort::None));
|
|
let err = cfg.validate().unwrap_err();
|
|
assert!(
|
|
err.contains("BUZZ_AGENT_THINKING_EFFORT=none"),
|
|
"error must name the value: {err}"
|
|
);
|
|
assert!(
|
|
err.contains("not valid for Anthropic"),
|
|
"error must name the provider: {err}"
|
|
);
|
|
assert!(
|
|
err.contains("low|medium|high|xhigh|max"),
|
|
"error must name allowed values: {err}"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn validate_rejects_minimal_effort_for_anthropic() {
|
|
let cfg = make_config_for_validation(Provider::Anthropic, Some(ThinkingEffort::Minimal));
|
|
let err = cfg.validate().unwrap_err();
|
|
assert!(err.contains("BUZZ_AGENT_THINKING_EFFORT=minimal"), "{err}");
|
|
assert!(err.contains("not valid for Anthropic"), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_all_efforts_for_databricks_v2() {
|
|
// DatabricksV2 dispatches across Anthropic/OpenAI/MLflow routes at request build time.
|
|
// No effort value is invalid for all three routes — startup rejects none.
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
] {
|
|
let cfg = make_config_for_validation(Provider::DatabricksV2, Some(effort));
|
|
assert!(
|
|
cfg.validate().is_ok(),
|
|
"DatabricksV2 must accept {effort:?} at startup (route-aware normalization at request build)"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_all_efforts_for_openai() {
|
|
// OpenAI effort support is model-dependent and normalized at request build time.
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
] {
|
|
let cfg = make_config_for_validation(Provider::OpenAi, Some(effort));
|
|
assert!(
|
|
cfg.validate().is_ok(),
|
|
"OpenAI must accept {effort:?} at startup (route-aware normalization at request build)"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_all_efforts_for_databricks() {
|
|
// Legacy Databricks effort support is model-dependent and normalized at request build time.
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
] {
|
|
let cfg = make_config_for_validation(Provider::Databricks, Some(effort));
|
|
assert!(
|
|
cfg.validate().is_ok(),
|
|
"Databricks must accept {effort:?} at startup (route-aware normalization at request build)"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_xhigh_for_anthropic() {
|
|
// xhigh is valid for Anthropic providers — model-level clamping is dynamic.
|
|
let cfg = make_config_for_validation(Provider::Anthropic, Some(ThinkingEffort::XHigh));
|
|
assert!(
|
|
cfg.validate().is_ok(),
|
|
"xhigh must be accepted at startup for Anthropic"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_max_for_anthropic() {
|
|
// max is valid for Anthropic providers.
|
|
let cfg = make_config_for_validation(Provider::Anthropic, Some(ThinkingEffort::Max));
|
|
assert!(cfg.validate().is_ok(), "max must be accepted for Anthropic");
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_xhigh_for_openai() {
|
|
// xhigh is valid for OpenAI providers (server-validated per-model).
|
|
let cfg = make_config_for_validation(Provider::OpenAi, Some(ThinkingEffort::XHigh));
|
|
assert!(cfg.validate().is_ok(), "xhigh must be accepted for OpenAI");
|
|
}
|
|
|
|
#[test]
|
|
fn validate_accepts_none_and_minimal_for_openai() {
|
|
// none/minimal are valid OpenAI effort values.
|
|
let cfg_none = make_config_for_validation(Provider::OpenAi, Some(ThinkingEffort::None));
|
|
assert!(
|
|
cfg_none.validate().is_ok(),
|
|
"none must be accepted for OpenAI"
|
|
);
|
|
let cfg_minimal =
|
|
make_config_for_validation(Provider::OpenAi, Some(ThinkingEffort::Minimal));
|
|
assert!(
|
|
cfg_minimal.validate().is_ok(),
|
|
"minimal must be accepted for OpenAI"
|
|
);
|
|
}
|
|
|
|
// ---- normalize_effort_for_openai_route ----
|
|
|
|
#[test]
|
|
fn normalize_openai_route_clamps_max_to_xhigh() {
|
|
// Use an unknown model so only the max→xhigh clamp fires, not per-model logic.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::Max, "llama-4"),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_passes_through_all_other_values_for_unknown_model() {
|
|
// Unknown/unverified models pass through unchanged (server-validated).
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
] {
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(effort, "unknown-future-model"),
|
|
effort,
|
|
"normalize_effort_for_openai_route must pass through {effort:?} for unknown model"
|
|
);
|
|
}
|
|
}
|
|
|
|
// ---- normalize_effort_for_anthropic_route ----
|
|
|
|
#[test]
|
|
fn normalize_anthropic_route_none_yields_none() {
|
|
assert_eq!(
|
|
normalize_effort_for_anthropic_route(ThinkingEffort::None),
|
|
None,
|
|
"none must yield None (omit thinking fields)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_anthropic_route_minimal_yields_none() {
|
|
assert_eq!(
|
|
normalize_effort_for_anthropic_route(ThinkingEffort::Minimal),
|
|
None,
|
|
"minimal must yield None (omit thinking fields)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_anthropic_route_passes_through_valid_values() {
|
|
for effort in [
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
] {
|
|
assert_eq!(
|
|
normalize_effort_for_anthropic_route(effort),
|
|
Some(effort),
|
|
"normalize_effort_for_anthropic_route must pass through {effort:?}"
|
|
);
|
|
}
|
|
}
|
|
|
|
// ---- F2: Fable 5 / Mythos 5 / Mythos Preview adaptive thinking ----
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_fable_5_emits_adaptive_and_effort() {
|
|
// Fable 5 — always-on adaptive thinking.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-fable-5", ThinkingEffort::High, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-fable-5");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-fable-5");
|
|
assert_eq!(oc["effort"], "high");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_mythos_5_emits_adaptive_and_effort() {
|
|
// Mythos 5 — always-on adaptive thinking.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-mythos-5", ThinkingEffort::Medium, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-mythos-5");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-mythos-5");
|
|
assert_eq!(oc["effort"], "medium");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_mythos_preview_emits_adaptive_and_effort() {
|
|
// Mythos Preview — Always on adaptive thinking.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-mythos-preview", ThinkingEffort::Low, 32_768);
|
|
let t = thinking.expect("thinking must be present for claude-mythos-preview");
|
|
assert_eq!(t["type"], "adaptive");
|
|
let oc = output_config.expect("output_config must be present for claude-mythos-preview");
|
|
assert_eq!(oc["effort"], "low");
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_passes_through_for_fable_5() {
|
|
// Fable 5 supports xhigh.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-fable-5", ThinkingEffort::XHigh),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_passes_through_for_mythos_5() {
|
|
// Mythos 5 supports xhigh.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-mythos-5", ThinkingEffort::XHigh),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_xhigh_clamped_to_high_for_mythos_preview() {
|
|
// Mythos Preview does NOT support xhigh — clamp to high.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-mythos-preview", ThinkingEffort::XHigh),
|
|
ThinkingEffort::High
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_fable_5() {
|
|
// Fable 5 supports max.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-fable-5", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_mythos_5() {
|
|
// Mythos 5 supports max.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-mythos-5", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn clamp_adaptive_effort_max_passes_through_for_mythos_preview() {
|
|
// Mythos Preview supports max.
|
|
assert_eq!(
|
|
clamp_adaptive_effort("claude-mythos-preview", ThinkingEffort::Max),
|
|
ThinkingEffort::Max
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_fable_5_xhigh_emits_xhigh() {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-fable-5", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
assert_eq!(output_config.unwrap()["effort"], "xhigh");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_mythos_5_xhigh_emits_xhigh() {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-mythos-5", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
assert_eq!(output_config.unwrap()["effort"], "xhigh");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_mythos_preview_xhigh_clamps_to_high() {
|
|
// Mythos Preview does NOT support xhigh → clamp to high.
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-mythos-preview", ThinkingEffort::XHigh, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
assert_eq!(
|
|
output_config.unwrap()["effort"],
|
|
"high",
|
|
"xhigh must clamp to high for claude-mythos-preview"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_fable_5_max_passes_through() {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-fable-5", ThinkingEffort::Max, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
assert_eq!(output_config.unwrap()["effort"], "max");
|
|
}
|
|
|
|
#[test]
|
|
fn anthropic_thinking_config_mythos_preview_max_passes_through() {
|
|
let (thinking, output_config) =
|
|
anthropic_thinking_config("claude-mythos-preview", ThinkingEffort::Max, 32_768);
|
|
let t = thinking.unwrap();
|
|
assert_eq!(t["type"], "adaptive");
|
|
assert_eq!(output_config.unwrap()["effort"], "max");
|
|
}
|
|
|
|
// ---- openai_efforts_for_model / normalize_effort_for_openai_route per-model table ----
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_gpt5_pro_high_only() {
|
|
// gpt-5-pro: high only — any other value must be substituted.
|
|
let supported = openai_efforts_for_model("gpt-5-pro").expect("gpt-5-pro must be in table");
|
|
assert_eq!(
|
|
supported,
|
|
&[ThinkingEffort::High],
|
|
"gpt-5-pro supports only high"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_gpt5_6_includes_max() {
|
|
let expected: &[ThinkingEffort] = &[
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
ThinkingEffort::Max,
|
|
];
|
|
|
|
for model in ["gpt-5.6", "gpt-5.6-sol", "gpt-5-6-sol", "goose-gpt-5-6-sol"] {
|
|
assert_eq!(
|
|
openai_efforts_for_model(model),
|
|
Some(expected),
|
|
"{model} must match the gpt-5.6 effort table"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_gpt5_5_includes_xhigh() {
|
|
let supported = openai_efforts_for_model("gpt-5.5").expect("gpt-5.5 must be in table");
|
|
assert!(
|
|
supported.contains(&ThinkingEffort::XHigh),
|
|
"gpt-5.5 must support xhigh"
|
|
);
|
|
assert!(
|
|
supported.contains(&ThinkingEffort::None),
|
|
"gpt-5.5 must support none"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_gpt5_1_excludes_xhigh_and_minimal() {
|
|
let supported = openai_efforts_for_model("gpt-5.1").expect("gpt-5.1 must be in table");
|
|
assert!(
|
|
!supported.contains(&ThinkingEffort::XHigh),
|
|
"gpt-5.1 must NOT support xhigh"
|
|
);
|
|
assert!(
|
|
!supported.contains(&ThinkingEffort::Minimal),
|
|
"gpt-5.1 must NOT support minimal"
|
|
);
|
|
assert!(
|
|
supported.contains(&ThinkingEffort::None),
|
|
"gpt-5.1 must support none"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_gpt5_base_excludes_none_includes_minimal() {
|
|
let supported = openai_efforts_for_model("gpt-5").expect("gpt-5 base must be in table");
|
|
assert!(
|
|
!supported.contains(&ThinkingEffort::None),
|
|
"gpt-5 base must NOT support none"
|
|
);
|
|
assert!(
|
|
supported.contains(&ThinkingEffort::Minimal),
|
|
"gpt-5 base must support minimal"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_unknown_returns_none() {
|
|
// Unknown models are not doc-verified — caller treats as server-validated pass-through.
|
|
assert!(openai_efforts_for_model("llama-4").is_none());
|
|
assert!(openai_efforts_for_model("claude-opus-4-8").is_none());
|
|
assert!(openai_efforts_for_model("gpt-4o").is_none());
|
|
}
|
|
|
|
// ---- Boundary-safe matching: version digits must not false-match longer versions ----
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_dated_base_ids_are_not_versioned() {
|
|
// gpt-5-1106: the "-1" is not version 5.1 — it's a date segment on the base model.
|
|
// Must fall through to base table, not gpt-5.1.
|
|
let result = openai_efforts_for_model("gpt-5-1106");
|
|
let base = openai_efforts_for_model("gpt-5").unwrap();
|
|
assert_eq!(
|
|
result,
|
|
Some(base),
|
|
"gpt-5-1106 must match base table (not gpt-5.1): got {result:?}"
|
|
);
|
|
// Crucially, must NOT support None (that's a gpt-5.1 property, not base).
|
|
assert!(
|
|
!result.unwrap().contains(&ThinkingEffort::None),
|
|
"gpt-5-1106 must NOT support none — base table only has minimal"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_gpt5_4o_is_base_not_5_4() {
|
|
// gpt-5-4o: the "-4" could false-match the gpt-5.4 family, but "4o" is a
|
|
// capability suffix on the base gpt-5 model, not version 5.4.
|
|
// Must fall through to base table.
|
|
let result = openai_efforts_for_model("gpt-5-4o");
|
|
let base = openai_efforts_for_model("gpt-5").unwrap();
|
|
assert_eq!(
|
|
result,
|
|
Some(base),
|
|
"gpt-5-4o must match base table (not gpt-5.4): got {result:?}"
|
|
);
|
|
// Crucially, must NOT support XHigh (that's a gpt-5.4 property, not base).
|
|
assert!(
|
|
!result.unwrap().contains(&ThinkingEffort::XHigh),
|
|
"gpt-5-4o must NOT support xhigh — that's a gpt-5.4 property and would 400"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_multi_digit_versions_pass_through() {
|
|
// Dotted two-digit versions (gpt-5.10, gpt5.10, gpt-5.50) must not match any known
|
|
// single-digit family — the digit boundary check on dotted tokens blocks them.
|
|
// These return None (server-validated pass-through).
|
|
assert!(
|
|
openai_efforts_for_model("gpt-5.10").is_none(),
|
|
"gpt-5.10 must pass through (unknown future model)"
|
|
);
|
|
assert!(
|
|
openai_efforts_for_model("gpt5.10").is_none(),
|
|
"gpt5.10 must pass through (unknown future model)"
|
|
);
|
|
assert!(
|
|
openai_efforts_for_model("gpt-5.50").is_none(),
|
|
"gpt-5.50 must pass through (not gpt-5.5)"
|
|
);
|
|
// Dash two-digit versions (gpt-5-10, databricks-gpt-5-10) look like short numeric
|
|
// version segments and must also pass through as unknown — not bucketed as base.
|
|
assert!(
|
|
openai_efforts_for_model("gpt-5-10").is_none(),
|
|
"gpt-5-10 must pass through (short numeric suffix = potential unrecognized version)"
|
|
);
|
|
assert!(
|
|
openai_efforts_for_model("databricks-gpt-5-10").is_none(),
|
|
"databricks-gpt-5-10 must pass through (short numeric suffix)"
|
|
);
|
|
// Short numeric suffix + textual continuation (e.g. a hypothetical 'gpt-5.10-preview')
|
|
// must also pass through — the digit count (1-3) determines version-like, regardless of
|
|
// what follows.
|
|
assert!(
|
|
openai_efforts_for_model("gpt-5-10-preview").is_none(),
|
|
"gpt-5-10-preview must pass through (short numeric version suffix with text tail)"
|
|
);
|
|
assert!(
|
|
openai_efforts_for_model("databricks-gpt-5-10-preview").is_none(),
|
|
"databricks-gpt-5-10-preview must pass through (short numeric version suffix with text tail)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_date_segment_with_suffix_is_base() {
|
|
// 4+ digit date segment followed by a textual suffix must still resolve to the base
|
|
// table — the date length (>=4) determines it's a build/date, not a version number.
|
|
let result = openai_efforts_for_model("gpt-5-1106-preview");
|
|
assert!(
|
|
result.is_some(),
|
|
"gpt-5-1106-preview must match base table (4-digit date segment)"
|
|
);
|
|
let supported = result.unwrap();
|
|
assert!(
|
|
supported.contains(&ThinkingEffort::Minimal),
|
|
"gpt-5-1106-preview (base) must support minimal"
|
|
);
|
|
assert!(
|
|
!supported.contains(&ThinkingEffort::None),
|
|
"gpt-5-1106-preview (base) must NOT support none"
|
|
);
|
|
assert!(
|
|
!supported.contains(&ThinkingEffort::XHigh),
|
|
"gpt-5-1106-preview (base) must NOT support xhigh"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_databricks_prefixed_still_matches() {
|
|
// Databricks-prefixed names (gateway forwarding) must still resolve to the right table.
|
|
let result = openai_efforts_for_model("databricks-gpt-5-5");
|
|
assert_eq!(
|
|
result,
|
|
openai_efforts_for_model("gpt-5.5"),
|
|
"databricks-gpt-5-5 must match gpt-5.5 family table"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_boundary_date_suffixed_still_matches() {
|
|
// Date-suffixed names (e.g. gpt-5.1-2025-04-01) must still resolve to the right family.
|
|
let result = openai_efforts_for_model("gpt-5.1-2025-04-01");
|
|
assert_eq!(
|
|
result,
|
|
openai_efforts_for_model("gpt-5.1"),
|
|
"gpt-5.1-2025-04-01 must match gpt-5.1 family table"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn openai_efforts_for_model_pro_before_base_gpt5() {
|
|
// gpt-5-pro must match the -pro table, not the base gpt-5 table.
|
|
let pro = openai_efforts_for_model("gpt-5-pro").unwrap();
|
|
let base = openai_efforts_for_model("gpt-5").unwrap();
|
|
assert_ne!(
|
|
pro, base,
|
|
"gpt-5-pro and gpt-5 base must hit different table entries"
|
|
);
|
|
assert_eq!(pro, &[ThinkingEffort::High]);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_pro_high_passes_through() {
|
|
// gpt-5-pro: high is the only supported value → high passes through unchanged.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::High, "gpt-5-pro"),
|
|
ThinkingEffort::High
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_pro_anything_but_high_becomes_high() {
|
|
// gpt-5-pro: any effort other than high must resolve to high.
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::XHigh,
|
|
] {
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(effort, "gpt-5-pro"),
|
|
ThinkingEffort::High,
|
|
"gpt-5-pro: {effort:?} must resolve to high"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_base_none_becomes_minimal() {
|
|
// gpt-5 base supports minimal but not none. none → minimal (peer fallback).
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::None, "gpt-5"),
|
|
ThinkingEffort::Minimal,
|
|
"gpt-5 base: none must fall back to minimal (peer)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_passes_max_through_for_gpt5_6() {
|
|
for model in ["gpt-5.6", "gpt-5-6-sol"] {
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::Max, model),
|
|
ThinkingEffort::Max,
|
|
"{model} must preserve max"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_5_max_becomes_xhigh() {
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::Max, "gpt-5.5"),
|
|
ThinkingEffort::XHigh,
|
|
"gpt-5.5 must clamp max to xhigh"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_5_minimal_becomes_none() {
|
|
// gpt-5.5 supports none but not minimal. minimal → none (peer fallback).
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::Minimal, "gpt-5.5"),
|
|
ThinkingEffort::None,
|
|
"gpt-5.5: minimal must fall back to none (peer)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_1_xhigh_becomes_high() {
|
|
// gpt-5.1 does not support xhigh → nearest supported below xhigh is high.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::XHigh, "gpt-5.1"),
|
|
ThinkingEffort::High,
|
|
"gpt-5.1: xhigh must resolve to high"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_4_xhigh_passes_through() {
|
|
// gpt-5.4 supports xhigh → pass through unchanged.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::XHigh, "gpt-5.4"),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_5_xhigh_passes_through() {
|
|
// gpt-5.5 supports xhigh → pass through unchanged.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::XHigh, "gpt-5.5"),
|
|
ThinkingEffort::XHigh
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_gpt5_dash_suffix_variants_match_correctly() {
|
|
// Databricks-prefixed or date-suffixed names must still hit the right family.
|
|
// "gpt-5.5" and "gpt-5-5" are treated identically; ditto for other families.
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::XHigh, "gpt-5-5"),
|
|
ThinkingEffort::XHigh,
|
|
"gpt-5-5 (dash) must match gpt-5.5 table"
|
|
);
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(ThinkingEffort::None, "gpt-5-1"),
|
|
ThinkingEffort::None,
|
|
"gpt-5-1 (dash) must match gpt-5.1 table"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn normalize_openai_route_unknown_model_passthrough() {
|
|
// Unknown models: all values pass through without substitution (server-validated).
|
|
for effort in [
|
|
ThinkingEffort::None,
|
|
ThinkingEffort::Minimal,
|
|
ThinkingEffort::Low,
|
|
ThinkingEffort::Medium,
|
|
ThinkingEffort::High,
|
|
ThinkingEffort::XHigh,
|
|
] {
|
|
assert_eq!(
|
|
normalize_effort_for_openai_route(effort, "llama-4"),
|
|
effort,
|
|
"unknown model: {effort:?} must pass through unchanged"
|
|
);
|
|
}
|
|
}
|
|
|
|
// ---- effort-table fixture sync guard ----------------------------------------
|
|
//
|
|
// Loads `effortTable.fixture.json` (the single source of truth shared with
|
|
// the TS test in `buzzAgentConfig.test.mjs`) and verifies that this Rust
|
|
// implementation produces the same valid-effort-value sets and default values
|
|
// as the TS `getProviderEffortConfig` function.
|
|
//
|
|
// Drift (a new model family added to one side but not the other) fails CI here
|
|
// before it can silently diverge in production.
|
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
|
|
/// Compute the valid effort values for a provider/model pair, mirroring
|
|
/// `getProviderEffortConfig` in `buzzAgentConfig.ts`.
|
|
///
|
|
/// Returns `(valid_values, default_value)` where `default_value` is `None`
|
|
/// for Anthropic manual-budget models (TS `defaultValue: null`), otherwise
|
|
/// `Some("medium")` or `Some("high")`.
|
|
fn valid_effort_values_for_provider_model(
|
|
provider: &str,
|
|
model: &str,
|
|
) -> (Vec<&'static str>, Option<&'static str>) {
|
|
const ALL_7: &[&str] = &["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
|
const ALL_EXCEPT_MAX: &[&str] = &["none", "minimal", "low", "medium", "high", "xhigh"];
|
|
const GPT5_PRO: &[&str] = &["high"];
|
|
const GPT5_1: &[&str] = &["none", "low", "medium", "high"];
|
|
|
|
let p = provider.to_ascii_lowercase();
|
|
// Strip arbitrary endpoint-naming prefix before model matching, mirroring TS and
|
|
// strip_catalog_prefix: find the first known family token (claude-, gpt-) and
|
|
// drop everything before it. Handles any catalog naming convention.
|
|
let raw_model = model.trim();
|
|
let lower_raw = raw_model.to_ascii_lowercase();
|
|
const FAMILY_TOKENS: &[&str] = &["claude-", "gpt-"];
|
|
let first_idx = FAMILY_TOKENS
|
|
.iter()
|
|
.filter_map(|tok| lower_raw.find(tok))
|
|
.min();
|
|
let stripped = match first_idx {
|
|
Some(idx) => &raw_model[idx..],
|
|
None => raw_model,
|
|
};
|
|
let m = stripped.to_ascii_lowercase();
|
|
|
|
// Thin adapter: converts production helper output to the string-based
|
|
// return type used by this function.
|
|
fn anthropic_result(m: &str) -> (Vec<&'static str>, Option<&'static str>) {
|
|
let (values, default) = anthropic_efforts_for_model(m);
|
|
let strs: Vec<&'static str> = values.iter().map(|e| e.openai_effort_str()).collect();
|
|
(strs, default.map(|e| e.openai_effort_str()))
|
|
}
|
|
|
|
fn openai_result(m: &str) -> (Vec<&'static str>, Option<&'static str>) {
|
|
if let Some(values) = openai_efforts_for_model(m) {
|
|
let strs: Vec<&'static str> =
|
|
values.iter().map(|e| e.openai_effort_str()).collect();
|
|
// Determine default from the family.
|
|
let default_val = if strs == GPT5_PRO {
|
|
Some("high")
|
|
} else if strs == GPT5_1 {
|
|
Some("none")
|
|
} else {
|
|
Some("medium")
|
|
};
|
|
(strs, default_val)
|
|
} else {
|
|
// Unknown model → all-except-max, default medium.
|
|
(ALL_EXCEPT_MAX.to_vec(), Some("medium"))
|
|
}
|
|
}
|
|
|
|
if p == "anthropic" {
|
|
return anthropic_result(&m);
|
|
}
|
|
if p == "openai" {
|
|
return openai_result(&m);
|
|
}
|
|
if p == "databricks_v2" {
|
|
if m.starts_with("claude-") {
|
|
return anthropic_result(&m);
|
|
}
|
|
// gpt-5 family check mirrors gpt5FamilyModel in TS.
|
|
let is_gpt5 = gpt5_token_matches(&m, "gpt-5-pro")
|
|
|| gpt5_token_matches(&m, "gpt5-pro")
|
|
|| gpt5_token_matches(&m, "gpt-5.6")
|
|
|| gpt5_token_matches(&m, "gpt5.6")
|
|
|| gpt5_token_matches(&m, "gpt-5-6")
|
|
|| gpt5_token_matches(&m, "gpt5-6")
|
|
|| gpt5_token_matches(&m, "gpt-5.5")
|
|
|| gpt5_token_matches(&m, "gpt5.5")
|
|
|| gpt5_token_matches(&m, "gpt-5.4")
|
|
|| gpt5_token_matches(&m, "gpt5.4")
|
|
|| gpt5_token_matches(&m, "gpt-5.1")
|
|
|| gpt5_token_matches(&m, "gpt5.1")
|
|
|| gpt5_base_matches(&m, "gpt-5")
|
|
|| gpt5_base_matches(&m, "gpt5");
|
|
if is_gpt5 {
|
|
return openai_result(&m);
|
|
}
|
|
if !m.is_empty() {
|
|
// Concrete non-claude, non-gpt5: MLflow path → all-except-max.
|
|
return openai_result(&m);
|
|
}
|
|
// Blank model: route unknown, all-7.
|
|
return (ALL_7.to_vec(), Some("medium"));
|
|
}
|
|
if p == "databricks" {
|
|
return openai_result(&m);
|
|
}
|
|
if p == "openrouter" {
|
|
return (ALL_7.to_vec(), Some("medium"));
|
|
}
|
|
// openai-compat, unknown, empty → all-7, default medium.
|
|
(ALL_7.to_vec(), Some("medium"))
|
|
}
|
|
|
|
#[derive(serde::Deserialize)]
|
|
struct FixtureEntry {
|
|
note: Option<String>,
|
|
provider: String,
|
|
model: String,
|
|
#[serde(rename = "validValues")]
|
|
valid_values: Vec<String>,
|
|
#[serde(rename = "defaultValue")]
|
|
default_value: Option<String>,
|
|
}
|
|
|
|
#[test]
|
|
fn effort_table_fixture_matches_rust_implementation() {
|
|
let fixture_json =
|
|
include_str!("../../../desktop/src/features/agents/ui/effortTable.fixture.json");
|
|
let entries: Vec<FixtureEntry> =
|
|
serde_json::from_str(fixture_json).expect("fixture must be valid JSON");
|
|
|
|
assert!(
|
|
!entries.is_empty(),
|
|
"fixture must contain at least one entry"
|
|
);
|
|
|
|
for entry in &entries {
|
|
let label = entry.note.as_deref().unwrap_or(entry.model.as_str());
|
|
let (valid_values, default_value) =
|
|
valid_effort_values_for_provider_model(&entry.provider, &entry.model);
|
|
|
|
let expected: Vec<&str> = entry.valid_values.iter().map(String::as_str).collect();
|
|
assert_eq!(
|
|
valid_values, expected,
|
|
"validValues mismatch for fixture entry \"{label}\" \
|
|
(provider={}, model={}): Rust side has {valid_values:?}, \
|
|
fixture expects {expected:?}",
|
|
entry.provider, entry.model,
|
|
);
|
|
|
|
let expected_default: Option<&str> = entry.default_value.as_deref();
|
|
assert_eq!(
|
|
default_value, expected_default,
|
|
"defaultValue mismatch for fixture entry \"{label}\" \
|
|
(provider={}, model={}): Rust side has {default_value:?}, \
|
|
fixture expects {expected_default:?}",
|
|
entry.provider, entry.model,
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_openrouter_with_key() {
|
|
assert_eq!(
|
|
resolve_provider(Some("openrouter"), None, None, Some("sk-or-123")).unwrap(),
|
|
Provider::OpenRouter
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn resolve_provider_openrouter_missing_key() {
|
|
let err = resolve_provider(Some("openrouter"), None, None, None).unwrap_err();
|
|
assert!(err.contains("OPENROUTER_API_KEY"));
|
|
}
|
|
}
|