mirror of
https://github.com/farion1231/cc-switch.git
synced 2026-07-24 12:44:18 +08:00
ac52c851bf
Mapped GPT models were rejected by Codex clients with "model does not support image inputs". Two root causes: - Catalog entries for native-Responses/Anthropic providers cloned a template whose input_modalities defaulted to ["text"], so every mapped model was advertised text-only. model_catalog_json replaces Codex's built-in model table wholesale, and both the TUI and the IDE extension block images pre-send when the current model is found without "image". - Editing the current Codex provider during proxy takeover only refreshed the DB backup, so removing the mapping left a stale model_catalog_json pointer (and its text-only catalog file) active in live config. Changes: - New shared model_capabilities module: explicit row declaration first, then a confirmed text-only registry (exact tail match only — prefix matching removed, variants enumerated so future -vl/-vision models fail open), everything else unknown. - Catalog generation writes input_modalities from that inference for all tool profiles: unknown models fail open to ["text","image"]; only confirmed text-only models are advertised as ["text"], giving users a clear client-side prompt instead of silent image stripping. - Live catalog reverse-import collapses modalities that match current inference, so registry corrections are not frozen into hidden row overrides and the rectifier's heuristic opt-out keeps working. - Saving the current Codex provider while takeover owns live now re-projects the live config (mirrors the hot-switch path), so mapping edits and removals take effect immediately. - Media rectifier delegates to the shared module; its preflight toggle is documented (4 locales) as proxy-request-only, never affecting catalog capability declarations.
277 lines
8.9 KiB
Rust
277 lines
8.9 KiB
Rust
use serde_json::Value;
|
|
|
|
/// Image-input capability shared by Codex catalog generation and proxy request
|
|
/// rectification.
|
|
///
|
|
/// `Unknown` is intentionally distinct from `Supported`: callers may choose
|
|
/// different execution policies without duplicating the model-name registry.
|
|
/// The Codex catalog treats unknown models as image-capable (fail open), while
|
|
/// the media rectifier leaves their request bodies untouched.
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
pub(crate) enum ImageInputCapability {
|
|
Supported,
|
|
Unsupported,
|
|
Unknown,
|
|
}
|
|
|
|
/// Resolve image-input capability from an explicit declaration first, then the
|
|
/// confirmed text-only model registry when the caller enables registry lookup.
|
|
pub(crate) fn resolve_image_input_capability(
|
|
model: &str,
|
|
declared_support: Option<bool>,
|
|
use_confirmed_registry: bool,
|
|
) -> ImageInputCapability {
|
|
match declared_support {
|
|
Some(true) => ImageInputCapability::Supported,
|
|
Some(false) => ImageInputCapability::Unsupported,
|
|
None if use_confirmed_registry && is_confirmed_text_only_model(model) => {
|
|
ImageInputCapability::Unsupported
|
|
}
|
|
None => ImageInputCapability::Unknown,
|
|
}
|
|
}
|
|
|
|
/// Resolve a model's image-input capability from the provider settings shapes
|
|
/// accepted by the proxy (`modelCatalog.models`, `modelCatalog`, or `models`).
|
|
pub(crate) fn image_input_capability_from_settings(
|
|
settings: &Value,
|
|
model: &str,
|
|
use_confirmed_registry: bool,
|
|
) -> ImageInputCapability {
|
|
resolve_image_input_capability(
|
|
model,
|
|
declared_model_image_support(settings, model),
|
|
use_confirmed_registry,
|
|
)
|
|
}
|
|
|
|
/// Convert a catalog row's explicit modality list into the shared capability
|
|
/// representation, falling back to the text-only registry when omitted.
|
|
pub(crate) fn image_input_capability_from_modalities(
|
|
model: &str,
|
|
modalities: Option<&[String]>,
|
|
) -> ImageInputCapability {
|
|
let declared_support = modalities.map(|items| {
|
|
items
|
|
.iter()
|
|
.any(|item| item.trim().eq_ignore_ascii_case("image"))
|
|
});
|
|
resolve_image_input_capability(model, declared_support, true)
|
|
}
|
|
|
|
/// Models that CC Switch is willing to advertise to clients as text-only.
|
|
///
|
|
/// This registry is deliberately exact and fail-open. A new suffix is not
|
|
/// inherited automatically: it remains image-capable until its capability is
|
|
/// confirmed, preventing a future `-vision`/`-vl` variant from being blocked by
|
|
/// the Codex client before a request can reach the proxy.
|
|
pub(crate) fn is_confirmed_text_only_model(model: &str) -> bool {
|
|
let normalized = normalize_model_id(model);
|
|
let tail = normalized.rsplit('/').next().unwrap_or(normalized.as_str());
|
|
|
|
const CONFIRMED_TAILS: &[&str] = &[
|
|
"ark-code-latest",
|
|
"deepseek-chat",
|
|
"deepseek-reasoner",
|
|
"deepseek-v4-flash",
|
|
"deepseek-v4-pro",
|
|
"glm-5.1",
|
|
// Exact rather than prefix matching: GLM visual models use a `v`
|
|
// suffix (for example glm-5.2v), which must remain image-capable.
|
|
"glm-5.2",
|
|
"kat-coder",
|
|
"kat-coder-pro",
|
|
"kat-coder-pro v1",
|
|
"kat-coder-pro v2",
|
|
"kat-coder-pro-v1",
|
|
"kat-coder-pro-v2",
|
|
"ling-2.5-1t",
|
|
"longcat-2.0",
|
|
"longcat-flash-chat",
|
|
"minimax-m2.7",
|
|
"minimax-m2.7-highspeed",
|
|
"mimo-v2.5-pro",
|
|
"qwen3-coder-480b",
|
|
"qwen3-coder-480b-a35b-instruct",
|
|
"qwen3-coder-flash",
|
|
"qwen3-coder-next",
|
|
"qwen3-coder-plus",
|
|
"step-3.5-flash",
|
|
"step-3.5-flash-2603",
|
|
"us.deepseek.r1-v1",
|
|
];
|
|
|
|
CONFIRMED_TAILS.contains(&tail)
|
|
}
|
|
|
|
fn declared_model_image_support(settings: &Value, model: &str) -> Option<bool> {
|
|
[
|
|
settings
|
|
.get("modelCatalog")
|
|
.and_then(|catalog| catalog.get("models")),
|
|
settings.get("modelCatalog"),
|
|
settings.get("models"),
|
|
]
|
|
.into_iter()
|
|
.flatten()
|
|
.find_map(|value| declared_model_image_support_in_value(value, model))
|
|
}
|
|
|
|
fn declared_model_image_support_in_value(value: &Value, model: &str) -> Option<bool> {
|
|
if let Some(models) = value.as_array() {
|
|
return models.iter().find_map(|entry| {
|
|
model_entry_matches(entry, None, model).then(|| explicit_image_support(entry))?
|
|
});
|
|
}
|
|
|
|
let object = value.as_object()?;
|
|
object.iter().find_map(|(key, entry)| {
|
|
model_entry_matches(entry, Some(key), model).then(|| explicit_image_support(entry))?
|
|
})
|
|
}
|
|
|
|
fn explicit_image_support(entry: &Value) -> Option<bool> {
|
|
if let Some(value) = entry
|
|
.get("supportsImage")
|
|
.or_else(|| entry.get("supports_image"))
|
|
.or_else(|| entry.get("vision"))
|
|
.and_then(Value::as_bool)
|
|
{
|
|
return Some(value);
|
|
}
|
|
|
|
[
|
|
entry.get("input"),
|
|
entry.pointer("/modalities/input"),
|
|
entry.get("input_modalities"),
|
|
entry.get("inputModalities"),
|
|
]
|
|
.into_iter()
|
|
.flatten()
|
|
.find_map(input_modalities_support_image)
|
|
}
|
|
|
|
fn input_modalities_support_image(value: &Value) -> Option<bool> {
|
|
let modalities = value.as_array()?;
|
|
Some(modalities.iter().any(|item| {
|
|
item.as_str()
|
|
.map(str::trim)
|
|
.is_some_and(|item| item.eq_ignore_ascii_case("image"))
|
|
}))
|
|
}
|
|
|
|
fn model_entry_matches(entry: &Value, key: Option<&str>, model: &str) -> bool {
|
|
key.is_some_and(|key| model_ids_match(key, model))
|
|
|| ["model", "id", "name"]
|
|
.into_iter()
|
|
.filter_map(|field| entry.get(field).and_then(Value::as_str))
|
|
.any(|candidate| model_ids_match(candidate, model))
|
|
}
|
|
|
|
fn model_ids_match(candidate: &str, model: &str) -> bool {
|
|
let candidate = normalize_model_id(candidate);
|
|
let model = normalize_model_id(model);
|
|
if candidate.is_empty() || model.is_empty() {
|
|
return false;
|
|
}
|
|
if candidate == model {
|
|
return true;
|
|
}
|
|
|
|
let candidate_tail = candidate.rsplit('/').next().unwrap_or(candidate.as_str());
|
|
let model_tail = model.rsplit('/').next().unwrap_or(model.as_str());
|
|
candidate_tail == model_tail || candidate == model_tail || candidate_tail == model
|
|
}
|
|
|
|
fn normalize_model_id(value: &str) -> String {
|
|
let mut normalized = value
|
|
.trim()
|
|
.trim_start_matches("models/")
|
|
.trim()
|
|
.to_ascii_lowercase();
|
|
if let Some(stripped) =
|
|
normalized.strip_suffix(crate::claude_desktop_config::ONE_M_CONTEXT_MARKER)
|
|
{
|
|
normalized = stripped.trim().to_string();
|
|
}
|
|
normalized
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use serde_json::json;
|
|
|
|
#[test]
|
|
fn gpt_and_unknown_models_remain_unknown_without_declarations() {
|
|
for model in ["gpt-5.4", "gpt-5.5", "gpt-5.6-sol", "custom-alias"] {
|
|
assert_eq!(
|
|
resolve_image_input_capability(model, None, true),
|
|
ImageInputCapability::Unknown,
|
|
"{model} must fail open"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn confirmed_text_only_registry_normalizes_namespaces_and_context_markers() {
|
|
assert!(is_confirmed_text_only_model("deepseek/deepseek-v4-pro"));
|
|
assert!(is_confirmed_text_only_model("GLM-5.2[1M]"));
|
|
assert!(is_confirmed_text_only_model("qwen/qwen3-coder-plus"));
|
|
assert!(is_confirmed_text_only_model(
|
|
"Qwen/Qwen3-Coder-480B-A35B-Instruct"
|
|
));
|
|
assert!(is_confirmed_text_only_model("MiniMax-M2.7-Highspeed"));
|
|
assert!(is_confirmed_text_only_model("step-3.5-flash-2603"));
|
|
assert!(!is_confirmed_text_only_model("glm-5.2v"));
|
|
}
|
|
|
|
#[test]
|
|
fn unconfirmed_family_suffixes_fail_open() {
|
|
for model in [
|
|
"minimax-m2.7-vision",
|
|
"qwen3-coder-ultra",
|
|
"qwen3-coder-vl",
|
|
"step-3.5-flash-vision",
|
|
] {
|
|
assert!(
|
|
!is_confirmed_text_only_model(model),
|
|
"unconfirmed variant {model} must not be hard-gated"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn explicit_capability_overrides_the_registry() {
|
|
assert_eq!(
|
|
resolve_image_input_capability("deepseek-v4-pro", Some(true), true),
|
|
ImageInputCapability::Supported
|
|
);
|
|
assert_eq!(
|
|
resolve_image_input_capability("gpt-5.4", Some(false), true),
|
|
ImageInputCapability::Unsupported
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn provider_settings_support_multiple_capability_shapes() {
|
|
let settings = json!({
|
|
"modelCatalog": {
|
|
"models": [
|
|
{ "model": "vision", "modalities": { "input": ["text", "image"] } },
|
|
{ "model": "text", "inputModalities": ["text"] }
|
|
]
|
|
}
|
|
});
|
|
|
|
assert_eq!(
|
|
image_input_capability_from_settings(&settings, "vision", true),
|
|
ImageInputCapability::Supported
|
|
);
|
|
assert_eq!(
|
|
image_input_capability_from_settings(&settings, "text", true),
|
|
ImageInputCapability::Unsupported
|
|
);
|
|
}
|
|
}
|