package config import ( "encoding/json" "os" "path/filepath" "strings" "testing" "time" "textmachine/backend/internal/llm" ) // models_catalog_test.go — the pack-12 point-1 catalog validation test (research/21 // §5.3, goose declarative.rs:396-467 `all_bundled_providers_are_valid`): a $0 go-test // over the SHIPPED configs/models.yaml that fails on catalog drift BEFORE a paid run. // LoadModels already fail-fasts most of this at load; the test PINS that the shipped // catalog currently passes (a stale prices_checked, an undeclared provider, a bad // reasoning/capability enum, a re-armed echo mine all turn this red), and adds the two // structural invariants LoadModels does not itself assert: every model-referenced // non-local provider declares an api_key_env, and every resolved min_max_tokens floor // is in a sane band. const shippedModelsYAML = "../../configs/models.yaml" // maxReasonableMinTokens bounds a per-model max_tokens floor. The real floors are // 8000 (DeepSeek/Gemini) and 16000 (Kimi); a value above this band is almost // certainly a typo (a token count written as a price, an extra zero) that would // over-reserve every call on that model. const maxReasonableMinTokens = 200000 func TestShippedModelsCatalogValid(t *testing.T) { m, err := LoadModels(shippedModelsYAML) if err != nil { // LoadModels aggregates ALL problems into one error — surface it verbatim so // the operator fixes the drift (stale prices, undeclared provider, bad enum, // echo mine) before spending a dollar. t.Fatalf("shipped configs/models.yaml failed validation (catalog drift before a paid run):\n%v", err) } // default_model resolves (price-fallback anchor). if _, ok := m.Models[m.DefaultModel]; !ok { t.Fatalf("default_model %q is not a defined model", m.DefaultModel) } knownReasoning := map[string]bool{"": true, "subset": true, "additive": true, "additive_total": true} for name, prov := range m.Providers { switch prov.Kind { case "openai", "anthropic", "local": default: t.Errorf("provider %s: unknown kind %q", name, prov.Kind) } if prov.Kind == "openai" && !knownReasoning[prov.Reasoning] { t.Errorf("provider %s: reasoning %q not in {subset,additive,additive_total}", name, prov.Reasoning) } } for name, mod := range m.Models { prov, ok := m.Providers[mod.Provider] if !ok { t.Errorf("model %s references undeclared provider %q", name, mod.Provider) continue } // Every model-reachable NON-LOCAL provider must declare an api_key_env: a paid // model with no key env would surface as a 401 only after reserve/slot charge. if prov.Kind != "local" && prov.APIKeyEnv == "" { t.Errorf("model %s: non-local provider %s declares no api_key_env", name, mod.Provider) } // Non-local models must carry non-zero input/output prices (an unknown model // must never book at $0 and blind the ceiling). if prov.Kind != "local" && (mod.Price.InputPerM <= 0 || mod.Price.OutputPerM <= 0) { t.Errorf("model %s: non-local model needs input/output prices > 0 (got in=%.4f out=%.4f)", name, mod.Price.InputPerM, mod.Price.OutputPerM) } // The RESOLVED max_tokens floor (provider→model layering) is sane: never // negative, never absurdly large. if floor := m.MinMaxTokens(name); floor < 0 || floor > maxReasonableMinTokens { t.Errorf("model %s: resolved min_max_tokens %d is out of the sane band [0,%d]", name, floor, maxReasonableMinTokens) } } } // TestShippedGeminiDeclaresOneSystemMessage is the gate the pack's own fix was missing: the two lines // in configs/models.yaml are the ONLY thing that makes the system-message join reach the real endpoint, // and an adversarial pass proved that deleting them left the entire module's tests green. Everything // else about the fix is exercised against a synthetic fixture provider; this is what ties it to // production. The failure it guards is a silent HTTP 200 with the glossary dropped — nothing 4xx, no // gate, a plausible translation that has quietly stopped obeying the memory bank. func TestShippedGeminiDeclaresOneSystemMessage(t *testing.T) { m, err := LoadModels(shippedModelsYAML) if err != nil { t.Fatalf("shipped configs/models.yaml failed validation:\n%v", err) } seen := 0 for name, mod := range m.Models { if mod.Provider != "gemini" { continue } seen++ if got := m.ResolveCapability(name).SystemMessages; got != llm.SystemMessagesSingle { t.Errorf("model %s rides the Gemini OpenAI-compat layer, which carries ONE system message; "+ "resolved SystemMessages = %q, want %q — the memory-bank injection is the second system "+ "message and is dropped by the endpoint without an error", name, got, llm.SystemMessagesSingle) } } if seen == 0 { t.Fatal("no model resolves to the gemini provider — the declaration this test guards has nothing to guard; " + "if the provider was removed on purpose, remove this test with it") } } // TestSystemMessagesIsRefusedOnTheAnthropicKind: the axis is an OpenAI-compat wire shape, and the // Anthropic adapter is built with no Capability at all — so a declaration there reaches nothing, while // still resolving into the capability the job snapshot carries. Accepted silently it would re-buy a // book for a line that changed no byte on the wire; refused by kind, the mistake is a load error. func TestSystemMessagesIsRefusedOnTheAnthropicKind(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "models.yaml") if err := os.WriteFile(path, []byte(` prices_checked: "`+time.Now().UTC().Format("2006-01-02")+`" default_model: m providers: a: kind: anthropic api_key_env: X capabilities: { system_messages: single } models: m: provider: a price: { input_per_m: 1, cached_per_m: 1, cache_write_per_m: 0, output_per_m: 1 } `), 0o644); err != nil { t.Fatal(err) } _, err := LoadModels(path) if err == nil { t.Fatal("a wire-shape axis the anthropic adapter cannot read must be refused, not silently resolved") } if !strings.Contains(err.Error(), "capabilities.system_messages") { t.Fatalf("the refusal must name the key, got: %v", err) } } // TestExplicitMultiCostsNothing pins the normalisation the code argues for: writing the DEFAULT out // loud is how an author records a verified endpoint or overrides a provider's single-slot rule, and it // must not put a key in the capability the snapshot carries — that would re-buy the book for a line // that changed no byte on the wire. func TestExplicitMultiCostsNothing(t *testing.T) { dir := t.TempDir() write := func(caps string) *Models { t.Helper() path := filepath.Join(dir, strings.ReplaceAll(caps, " ", "")+"models.yaml") if err := os.WriteFile(path, []byte(` prices_checked: "`+time.Now().UTC().Format("2006-01-02")+`" default_model: m providers: p: kind: openai base_url: http://x api_key_env: X `+caps+` models: m: provider: p price: { input_per_m: 1, cached_per_m: 1, cache_write_per_m: 0, output_per_m: 1 } `), 0o644); err != nil { t.Fatal(err) } m, err := LoadModels(path) if err != nil { t.Fatalf("load: %v", err) } return m } bare := write("") multi := write(" capabilities: { system_messages: multi }\n") single := write(" capabilities: { system_messages: single }\n") bareJSON, _ := json.Marshal(bare.ResolveCapability("m")) multiJSON, _ := json.Marshal(multi.ResolveCapability("m")) singleJSON, _ := json.Marshal(single.ResolveCapability("m")) if string(bareJSON) != string(multiJSON) { t.Fatalf("an explicit `multi` must resolve byte-identically to saying nothing:\n bare %s\n multi %s", bareJSON, multiJSON) } if string(singleJSON) == string(bareJSON) { t.Fatalf("a declared `single` MUST move the snapshot bytes — that is what makes the flip a loud --resnapshot: %s", singleJSON) } } // TestAProviderThatWillWaitLongSaysSoInTheCatalog is the second half of the derived deadline, and it // is the half that answers the standing review question — «does a provider the repository has never // seen work without editing Go». // // It does, and this is the price: a provider whose models declare a big enough max_tokens FLOOR is // declaring that its calls are big, and a big call under the vendor DEFAULT speed derives a deadline // far past whatever attempt_s the config carries. That is not a failure — the derivation is // deliberately slower than any provider we have measured, so it only ever grants MORE time — but it // is a quarter of an hour of waiting that the config never mentions, and an operator who reads // `attempt_s: 240` and watches a call run for fifteen minutes has been told something untrue by his // own configuration. // // So the catalogue has to say ONE of two things about such a provider, and either is a line of DATA: // - `tok_s_floor` — the speed it actually holds, measured from its own request_log; or // - `attempt_max_s` — how long we are willing to wait for it, which is a policy and needs no measurement. // // ⛔ IT DOES NOT DEMAND THE MEASUREMENT. Requiring `tok_s_floor` would force a number to be invented // for every provider nobody has run yet, and a fabricated floor is worse than an honest default: it // would cut real calls we had already paid for. The alternative is the point. func TestAProviderThatWillWaitLongSaysSoInTheCatalog(t *testing.T) { m, err := LoadModels(shippedModelsYAML) if err != nil { t.Fatalf("load: %v", err) } // The largest budget one call to each provider can carry: its biggest declared floor, doubled once // by a regeneration. It is a LOWER bound on what a real run may ask for (a long chunk derives more // than the floor), which is the conservative direction for a gate: it can only under-report who // needs a line, never demand one nobody needs. biggest := map[string]int{} for name, mod := range m.Models { if floor := m.MinMaxTokens(name); floor*2 > biggest[mod.Provider] { biggest[mod.Provider] = floor * 2 } } checked, flagged := 0, 0 for prov, grant := range biggest { if grant == 0 { continue // no model of this provider declares a floor: nothing to say } p := m.Providers[prov] derived := p.Timeouts.Profile().DeadlineFor(grant) configured := time.Duration(p.Timeouts.AttemptS) * time.Second if derived <= configured { continue // every call fits inside the deadline the config already names } checked++ if p.Timeouts.TokSFloor <= 0 && p.Timeouts.AttemptMaxS <= 0 { flagged++ t.Errorf("provider %q can be asked for %d output tokens, which derives a %s deadline against "+ "its configured attempt_s of %s — declare timeouts.tok_s_floor (measure it from this "+ "provider's own request_log) or timeouts.attempt_max_s (how long we are willing to wait)", prov, grant, derived, configured) } } // The control beside the negative: «0 providers need a line» and «the loop never ran» print the // same on a green test, and only one of them means anything. t.Logf("providers in the catalogue: %d; with a declared token floor: %d; whose biggest call outgrows "+ "its own attempt_s (and therefore must declare one of the two fields): %d; missing both: %d", len(m.Providers), len(biggest), checked, flagged) if checked == 0 { t.Fatalf("no provider in the shipped catalogue outgrows its own attempt_s — this gate is asking a " + "question of nothing, and would stay green through any regression in the derivation") } } // TestEveryDeclaredDeadlineKnobStaysDeclared pins the deadline data as data. Each of these numbers was // PAID FOR — a speed floor is the p10 of that provider's own request_log over hundreds of rows, a queue // slack is the vendor's published figure — and none of them is reachable by the gate above: it walks the // LOADED struct, where a deleted line and a never-declared field are the same zero, and it skips a // provider whose models declare no token floor at all. So deleting `tok_s_floor: 35` from zai passed // every case in this package. // // The consequence of losing one is quiet by construction. A missing speed floor falls back to the vendor // default, which is within two percent of zai's measured value — the run does not break, the measurement // is simply gone, and the next person re-derives it from scratch or does without. func TestEveryDeclaredDeadlineKnobStaysDeclared(t *testing.T) { m, err := LoadModels(shippedModelsYAML) if err != nil { t.Fatalf("load: %v", err) } type knobs struct { tokSFloor float64 queueSlackS int attemptMaxS int } // The catalogue as it ships. A change here is a change to a measured or vendor-published number and // must be made deliberately, with the measurement said beside it in models.yaml. want := map[string]knobs{ "deepseek": {tokSFloor: 50, queueSlackS: 600, attemptMaxS: 1240}, "zai": {tokSFloor: 35}, "kimi": {attemptMaxS: 1200}, "gemini": {attemptMaxS: 1200}, } declared := 0 for name, p := range m.Providers { got := knobs{p.Timeouts.TokSFloor, p.Timeouts.QueueSlackS, p.Timeouts.AttemptMaxS} if got != (knobs{}) { declared++ } w, pinned := want[name] if !pinned { if got != (knobs{}) { t.Errorf("provider %q declares deadline knobs %+v that this pin does not know about — add "+ "them here with the measurement that produced them, or the next edit loses them silently", name, got) } continue } if got != w { t.Errorf("provider %q: deadline knobs moved from %+v to %+v. These are measured numbers (a "+ "speed floor is the p10 of this provider's own request_log) — losing one returns the "+ "provider to the vendor default without a word anywhere", name, w, got) } } // The control beside the count: «every knob is where it was» and «the catalogue declares none» print // the same on a green test. if declared != len(want) { t.Fatalf("premise broken: %d providers declare deadline knobs, the pin names %d — the two must be "+ "the same set or one side is not being read", declared, len(want)) } t.Logf("providers in the catalogue: %d; declaring deadline knobs: %d, all pinned", len(m.Providers), declared) } // TestACapBelowTheFloorIsRefusedAtLoad pins the one relation between the deadline knobs that no // per-field check can see. The derivation clamps UP to `attempt_s` and then DOWN to `attempt_max_s`, so // a cap below the floor wins: every call to that provider quietly gets less time than the file declares, // and `attempt_s`'s own doccomment — which calls it a FLOOR — stops being true. Nothing logs it; the // calls simply come back cut. func TestACapBelowTheFloorIsRefusedAtLoad(t *testing.T) { load := func(timeouts string) error { t.Helper() body := "prices_checked: " + time.Now().UTC().Format("2006-01-02") + ` default_model: fake providers: p: { kind: openai, base_url: http://x, timeouts: ` + timeouts + ` } models: fake: { provider: p, price: { input_per_m: 1, output_per_m: 2 } } ` _, err := LoadModels(writeTmp(t, filepath.Join(t.TempDir(), "models.yaml"), body)) return err } // The control FIRST: a cap ABOVE the floor is an ordinary configuration and must load, or the check // below would be satisfied by a loader that refuses everything. if err := load("{ attempt_s: 240, max_attempts: 2, attempt_max_s: 1200 }"); err != nil { t.Fatalf("a cap above the floor is a legitimate configuration and must load: %v", err) } if err := load("{ attempt_s: 240, max_attempts: 2 }"); err != nil { t.Fatalf("an unset cap means «the derivation stands on its own» and must load: %v", err) } err := load("{ attempt_s: 240, max_attempts: 2, attempt_max_s: 120 }") if err == nil { t.Fatal("a cap BELOW the floor must be refused at load: the cap wins in the clamp, so every call " + "gets less time than the declared floor and nothing anywhere says so") } if !strings.Contains(err.Error(), "attempt_max_s") || !strings.Contains(err.Error(), "attempt_s") { t.Fatalf("the refusal must name BOTH knobs, or the operator cannot see which pair is wrong: %v", err) } // And the typo guard, which is the other half of the same field's danger: a slipped digit turns a // ten-minute wait into an afternoon, and the only place it shows is a run that looks hung. if err := load("{ attempt_s: 240, max_attempts: 2, queue_slack_s: 600000 }"); err == nil { t.Fatal("a queue slack of a week must be refused as a typo — nobody chose to wait that long") } }