Add Together AI platform + bare-array listing tolerance (provider 9)

The 12th registry row: id "together", TOGETHER_API_KEY > auth.json
"together" scope, https://api.together.xyz/v1 with KIGI_TOGETHER_BASE_URL
override, Bearer, ChatCompletions + Passthrough, enrichment-backed
metadata (models_dev_id togetherai) with the tool-calling listing
restriction (Together's listing mixes chat/embedding/rerank/image types).

Together's GET /v1/models returns a BARE JSON ARRAY, not the OpenAI
{object:list,data:[]} envelope. New shared parser parse_openai_listing
tolerates both shapes (sniffs the top-level [ vs { for accurate
diagnostics); every OpenAI-listing provider now routes through it, with
byte-equivalent envelope behavior (verified: Groq/Google/OpenRouter
unchanged) and non-silent errors. Together's org/Model ids match the
togetherai snapshot keys exactly (verified), so restrict_to_enriched keeps
the ~27 tool-calling models without the id-shape trap.

Review: ship-ready, shared-parser change proven strictly-additive and safe,
registry/e2e/dialect all correct. Backlog logged: kigi ignores the wire
per-model  field, so a live Together chat model absent from
models.dev is dropped until indexed — a future wire- filter would
unlock the fresh full catalog.
This commit is contained in:
2026-07-21 13:16:20 -04:00
parent e7dbb98207
commit 0a147bd8a5
4 changed files with 190 additions and 10 deletions
@@ -603,7 +603,8 @@ mod tests {
"mistral",
"fireworks",
"google",
"openrouter"
"openrouter",
"together"
]
);
assert_eq!(default_id(&built), Some(XAI_API_KEY_METHOD_ID));
@@ -637,7 +638,8 @@ mod tests {
"mistral",
"fireworks",
"google",
"openrouter"
"openrouter",
"together"
]
);
assert_eq!(default_id(&built), Some(CACHED_TOKEN_AUTH_METHOD_ID));
@@ -664,7 +666,8 @@ mod tests {
"mistral",
"fireworks",
"google",
"openrouter"
"openrouter",
"together"
]
);
assert_eq!(default_id(&built), Some(CACHED_TOKEN_AUTH_METHOD_ID));
@@ -694,7 +697,8 @@ mod tests {
"mistral",
"fireworks",
"google",
"openrouter"
"openrouter",
"together"
]
);
assert_eq!(default_id(&built), None);
@@ -297,7 +297,13 @@ fn fetch_one_platform_models(
.map(|s| s.to_string());
let data = match platform.listing() {
kigi_models::ListingDialect::OpenAi => {
response.json::<kigi_models::WireModelsResponse>()?.data
// Tolerant of both the {data:[...]} envelope and a bare array
// (Together AI serves the bare form).
let body = response.text()?;
kigi_models::parse_openai_listing(&body).map_err(|e| BackendError::RequestFailed {
status: 200,
body: format!("openai listing parse failed: {e}"),
})?
}
kigi_models::ListingDialect::Anthropic => {
let body = response.text()?;
@@ -1597,6 +1603,98 @@ mod tests {
);
}
/// Together-cycle e2e: the listing is a BARE JSON ARRAY (no {data:[]}
/// envelope) — parse_openai_listing tolerates it. models.dev enrichment
/// (matching org/Model keys) supplies context + the tool-calling
/// restriction drops non-chat models; Passthrough dialect; slashed id.
#[tokio::test(flavor = "multi_thread")]
#[serial_test::serial]
async fn together_bare_array_listing_enriches_restricts_and_maps_passthrough() {
let platform_server = wiremock::MockServer::start().await;
wiremock::Mock::given(wiremock::matchers::method("GET"))
.and(wiremock::matchers::path("/models"))
.and(wiremock::matchers::header("Authorization", "Bearer tg-1"))
.respond_with(wiremock::ResponseTemplate::new(200).set_body_json(
// BARE ARRAY, not {object:list,data:[]}.
serde_json::json!([
{ "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "object": "model",
"type": "chat", "context_length": 262144 },
{ "id": "togethercomputer/m2-bert-80M-8k-retrieval", "object": "model",
"type": "embedding" }
]),
))
.expect(1)
.mount(&platform_server)
.await;
let modelsdev_server = wiremock::MockServer::start().await;
wiremock::Mock::given(wiremock::matchers::method("GET"))
.and(wiremock::matchers::path("/api.json"))
.respond_with(wiremock::ResponseTemplate::new(200).set_body_json(
serde_json::json!({ "togetherai": { "models": {
"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
"limit": {"context": 262144, "output": 32768},
"tool_call": true
},
"togethercomputer/m2-bert-80M-8k-retrieval": { "limit": {"context": 8192} }
}}}),
))
.expect(1)
.mount(&modelsdev_server)
.await;
let cache_dir = tempfile::tempdir().unwrap();
let _base = kigi_test_support::EnvGuard::set(
kigi_models::TOGETHER_BASE_URL_ENV,
platform_server.uri(),
);
let _mdev = kigi_test_support::EnvGuard::set(
crate::agent::enrichment_fetch::MODELS_DEV_URL_ENV,
format!("{}/api.json", modelsdev_server.uri()),
);
let _mdev_cache = kigi_test_support::EnvGuard::set(
crate::agent::enrichment_fetch::MODELS_DEV_CACHE_DIR_ENV,
cache_dir.path(),
);
let endpoints = crate::agent::config::EndpointsConfig::default();
let keys = crate::agent::models::PlatformApiKeys::test_single(
kigi_models::PlatformId::Together,
"tg-1",
);
let result = tokio::task::spawn_blocking(move || {
fetch_platform_models_blocking(&endpoints, None, &keys)
})
.await
.unwrap()
.expect("fetch must succeed");
assert_eq!(
result
.models
.iter()
.map(|m| m.id.as_deref().unwrap_or_default())
.collect::<Vec<_>>(),
vec!["together/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8"],
"bare array parsed; embedding (not tool-calling) dropped; slashed id in key"
);
let entry = &result.models[0];
assert_eq!(entry.context_window.get(), 262_144);
assert_eq!(entry.max_completion_tokens, Some(32_768));
assert_eq!(
entry.model, "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
"the native slashed id rides the wire"
);
let model_entry = crate::agent::config::ModelEntry::from_config_entry(entry);
let creds = crate::agent::config::ResolvedCredentials {
api_key: Some("tg-1".into()),
base_url: entry.base_url.clone(),
auth_type: kigi_chat_state::AuthType::ApiKey,
auth_scheme: Default::default(),
};
let cfg = crate::agent::config::sampling_config_for_model(&model_entry, creds, None);
assert_eq!(
cfg.chat_compat,
kigi_sampling_types::ChatCompat::Passthrough
);
}
#[test]
fn get_env_keys_parses_strings_and_rejects_non_strings() {
use crate::agent::config::EnvKeys;