Add per-model capability wire-shape layer (D3.1) and fold context-assembly, memory-version, and resolved capability into the job snapshot (D5.2)
This commit is contained in:
parent
dee6efa282
commit
636919c52f
13 changed files with 876 additions and 68 deletions
|
|
@ -36,6 +36,16 @@ providers:
|
||||||
reasoning: subset
|
reasoning: subset
|
||||||
timeouts: { attempt_s: 240, max_attempts: 3, backoff_cap_s: 60 }
|
timeouts: { attempt_s: 240, max_attempts: 3, backoff_cap_s: 60 }
|
||||||
|
|
||||||
|
xai: # Grok — центральная модель Фазы 1: дефолт-редактор SFW + главный тир канала B + судья 18+.
|
||||||
|
# ⚠️ ГИГИЕНА АККАУНТОВ (D8/оркестратор): XAI_API_KEY обязан быть ЧИСТЫМ прод-аккаунтом
|
||||||
|
# БЕЗ data-sharing. Промо-$150-аккаунт (data-sharing) — ТОЛЬКО eval/NSFW, реальные книги
|
||||||
|
# через него гнать нельзя. По ключу это не проверяется — подтвердить перед боевым прогоном.
|
||||||
|
kind: openai
|
||||||
|
base_url: https://api.x.ai/v1
|
||||||
|
api_key_env: XAI_API_KEY
|
||||||
|
reasoning: additive # xAI биллит reasoning ПОВЕРХ completion (проверено vojo); think-ON заблокирован до резерва в EstimateUSD (D6.2/F3)
|
||||||
|
timeouts: { attempt_s: 240, max_attempts: 3, backoff_cap_s: 60 }
|
||||||
|
|
||||||
# Anthropic убран из стека по решению владельца (04.07): дорого, в RU-контуре
|
# Anthropic убран из стека по решению владельца (04.07): дорого, в RU-контуре
|
||||||
# экономически недоступен (Р5), в eval не валидирован. Судья Р4 — Gemini
|
# экономически недоступен (Р5), в eval не валидирован. Судья Р4 — Gemini
|
||||||
# (нативный адаптер, Фаза 2). Нативный Anthropic-адаптер в коде помечен
|
# (нативный адаптер, Фаза 2). Нативный Anthropic-адаптер в коде помечен
|
||||||
|
|
@ -60,14 +70,41 @@ models:
|
||||||
glm-5:
|
glm-5:
|
||||||
provider: zai
|
provider: zai
|
||||||
price: { input_per_m: 1.0, cached_per_m: 0.2, cache_write_per_m: 0, output_per_m: 3.2 }
|
price: { input_per_m: 1.0, cached_per_m: 0.2, cache_write_per_m: 0, output_per_m: 3.2 }
|
||||||
extra_body: { thinking: { type: disabled } } # эмпирика полигона: thinking GLM — таймауты ×3
|
# thinking-выключатель переехал из extra_body в capabilities.reasoning (D3.1):
|
||||||
|
# тело запроса байт-в-байт то же ({thinking:{type:disabled}} при reasoning=off),
|
||||||
|
# но теперь это единая wire-форма — и off_extra_body разблокирует think-ON
|
||||||
|
# эскалацию (D6.2) без второй ручки. Смена снапшота осознанная (--resnapshot).
|
||||||
|
capabilities:
|
||||||
|
reasoning:
|
||||||
|
control: extra_body_disable
|
||||||
|
off_extra_body: { thinking: { type: disabled } } # эмпирика полигона: thinking GLM — таймауты ×3
|
||||||
note: Редактор ru-стиля (Р4). Флагманы линейки уже GLM-5.1/5.2 ($1.4/$4.4) — пересмотреть к Фазе 2.
|
note: Редактор ru-стиля (Р4). Флагманы линейки уже GLM-5.1/5.2 ($1.4/$4.4) — пересмотреть к Фазе 2.
|
||||||
|
|
||||||
kimi-k2.6:
|
kimi-k2.6:
|
||||||
provider: kimi
|
provider: kimi
|
||||||
price: { input_per_m: 0.95, cached_per_m: 0.16, cache_write_per_m: 0, output_per_m: 4.0 }
|
price: { input_per_m: 0.95, cached_per_m: 0.16, cache_write_per_m: 0, output_per_m: 4.0 }
|
||||||
|
# Альт-редактор (bake-off эксп-04, проиграл grok-4.3): принимает ТОЛЬКО temperature:1
|
||||||
|
# (иначе 400), max_tokens≥16000 (иначе content пустой с finish=length), ~11k reasoning
|
||||||
|
# tok/абзац. capabilities фиксируют квирк, чтобы edit-стадия на temp 0.4 не 400-ла.
|
||||||
|
capabilities:
|
||||||
|
temperature: { mode: force, value: 1 }
|
||||||
note: Альтернативный редактор (Р4).
|
note: Альтернативный редактор (Р4).
|
||||||
|
|
||||||
|
grok-4.3:
|
||||||
|
provider: xai
|
||||||
|
# $1.25/$2.50/1M (факт. 2026-07-04, docs.x.ai). cached_per_m=input: кэш-цену xAI
|
||||||
|
# не верифицировали, а для редактора (черновик — свежий вход на чанк) cache-hit≈0,
|
||||||
|
# поэтому консервативно не моделируем скидку — потолок не слепнет. Уточнит полигон.
|
||||||
|
price: { input_per_m: 1.25, cached_per_m: 1.25, cache_write_per_m: 0, output_per_m: 2.50 }
|
||||||
|
# Дефолт-редактор (bake-off эксп-04): thinking OFF (0 reasoning-токенов, ~13с), temperature шлём.
|
||||||
|
capabilities:
|
||||||
|
temperature: { mode: send }
|
||||||
|
reasoning: { control: none }
|
||||||
|
note: >-
|
||||||
|
Дефолт-редактор SFW (D3, bake-off эксп-04) + главный пермиссив-тир канала B + судья 18+.
|
||||||
|
reasoning additive: think-ON только после резерва в EstimateUSD (D6.2/техдолг F3). Заведена,
|
||||||
|
в стадию не включена — переключение editor→grok в следующем конфиг-шаге (после подтверждения прод-ключа).
|
||||||
|
|
||||||
# claude-sonnet-5 / claude-opus-4-8 удалены вместе с провайдером anthropic
|
# claude-sonnet-5 / claude-opus-4-8 удалены вместе с провайдером anthropic
|
||||||
# (см. комментарий у providers выше). Судья Фазы 2 — Gemini (нативный адаптер).
|
# (см. комментарий у providers выше). Судья Фазы 2 — Gemini (нативный адаптер).
|
||||||
|
|
||||||
|
|
|
||||||
132
backend/internal/config/capability_test.go
Normal file
132
backend/internal/config/capability_test.go
Normal file
|
|
@ -0,0 +1,132 @@
|
||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"textmachine/backend/internal/llm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestResolveCapabilityMerge checks the D3.1 layering: kind baseline < provider
|
||||||
|
// default < per-model override, field-by-field with the model winning.
|
||||||
|
func TestResolveCapabilityMerge(t *testing.T) {
|
||||||
|
m := &Models{
|
||||||
|
Providers: map[string]Provider{
|
||||||
|
"openai": {Kind: "openai", Capabilities: &CapabilitiesConfig{
|
||||||
|
BudgetField: "max_completion_tokens",
|
||||||
|
Temperature: &TemperatureCap{Mode: "omit"},
|
||||||
|
}},
|
||||||
|
"local": {Kind: "local"},
|
||||||
|
},
|
||||||
|
Models: map[string]Model{
|
||||||
|
"gpt-5-nano": {Provider: "openai"}, // inherits the provider default
|
||||||
|
"gpt-5-mini": {Provider: "openai", Capabilities: &CapabilitiesConfig{
|
||||||
|
Temperature: &TemperatureCap{Mode: "force", Value: 0.7}, // overrides temp only
|
||||||
|
}},
|
||||||
|
"local-x": {Provider: "local"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
nano := m.ResolveCapability("gpt-5-nano")
|
||||||
|
if nano.Budget != llm.BudgetMaxCompletionTokens || nano.Temp != llm.TempOmit {
|
||||||
|
t.Fatalf("nano must inherit provider default: %+v", nano)
|
||||||
|
}
|
||||||
|
|
||||||
|
mini := m.ResolveCapability("gpt-5-mini")
|
||||||
|
if mini.Budget != llm.BudgetMaxCompletionTokens {
|
||||||
|
t.Fatalf("mini budget must still inherit the provider: %+v", mini)
|
||||||
|
}
|
||||||
|
if mini.Temp != llm.TempForce || mini.TempValue != 0.7 {
|
||||||
|
t.Fatalf("mini temperature override must win: %+v", mini)
|
||||||
|
}
|
||||||
|
|
||||||
|
loc := m.ResolveCapability("local-x")
|
||||||
|
if loc.Reasoning.Control != llm.ReasoningEffortField || loc.Reasoning.OffEffort != "none" {
|
||||||
|
t.Fatalf("local baseline must map off->none: %+v", loc.Reasoning)
|
||||||
|
}
|
||||||
|
if loc.Budget != llm.BudgetMaxTokens || loc.Temp != llm.TempSend {
|
||||||
|
t.Fatalf("local baseline budget/temp: %+v", loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLoadModelsRejectsBadCapability confirms a bogus capability enum fails
|
||||||
|
// fast at load (part of the models.yaml problem list), not as a runtime 4xx.
|
||||||
|
func TestLoadModelsRejectsBadCapability(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
path := filepath.Join(dir, "models.yaml")
|
||||||
|
body := fmt.Sprintf(`
|
||||||
|
prices_checked: %q
|
||||||
|
default_model: m
|
||||||
|
providers:
|
||||||
|
p:
|
||||||
|
kind: openai
|
||||||
|
base_url: http://x
|
||||||
|
models:
|
||||||
|
m:
|
||||||
|
provider: p
|
||||||
|
price: { input_per_m: 1, output_per_m: 2 }
|
||||||
|
capabilities:
|
||||||
|
budget_field: bogus
|
||||||
|
`, time.Now().UTC().Format("2006-01-02"))
|
||||||
|
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err := LoadModels(path)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "budget_field") {
|
||||||
|
t.Fatalf("bad budget_field must fail-fast mentioning the field, got %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLoadModelsCapabilityValidation covers the other validateCapabilities
|
||||||
|
// paths the single case above misses: the PROVIDER call-site, the
|
||||||
|
// temperature.mode / reasoning.control enums, and companion-field completeness
|
||||||
|
// (a control whose disable field is absent must fail at load, not no-op at
|
||||||
|
// runtime).
|
||||||
|
func TestLoadModelsCapabilityValidation(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
capsYAML string // injected under provider p or model m as noted
|
||||||
|
onProvider bool
|
||||||
|
wantSubstr string
|
||||||
|
}{
|
||||||
|
{"provider bad reasoning.control", "capabilities: { reasoning: { control: bogus } }", true, "control"},
|
||||||
|
{"model bad temperature.mode", "capabilities: { temperature: { mode: sned } }", false, "mode"},
|
||||||
|
{"extra_body_disable without off_extra_body", "capabilities: { reasoning: { control: extra_body_disable } }", false, "off_extra_body"},
|
||||||
|
{"effort without off_effort", "capabilities: { reasoning: { control: effort } }", false, "off_effort"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
provCaps, modelCaps := "", ""
|
||||||
|
if c.onProvider {
|
||||||
|
provCaps = "\n " + c.capsYAML
|
||||||
|
} else {
|
||||||
|
modelCaps = "\n " + c.capsYAML
|
||||||
|
}
|
||||||
|
body := fmt.Sprintf(`
|
||||||
|
prices_checked: %q
|
||||||
|
default_model: m
|
||||||
|
providers:
|
||||||
|
p:
|
||||||
|
kind: openai
|
||||||
|
base_url: http://x%s
|
||||||
|
models:
|
||||||
|
m:
|
||||||
|
provider: p
|
||||||
|
price: { input_per_m: 1, output_per_m: 2 }%s
|
||||||
|
`, time.Now().UTC().Format("2006-01-02"), provCaps, modelCaps)
|
||||||
|
dir := t.TempDir()
|
||||||
|
path := filepath.Join(dir, "models.yaml")
|
||||||
|
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err := LoadModels(path)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), c.wantSubstr) {
|
||||||
|
t.Fatalf("expected fail-fast mentioning %q, got %v", c.wantSubstr, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -44,6 +44,34 @@ type Provider struct {
|
||||||
// расстроила бы локальную модель. 0 = наследовать запрос.
|
// расстроила бы локальную модель. 0 = наследовать запрос.
|
||||||
Temperature float64 `yaml:"temperature"`
|
Temperature float64 `yaml:"temperature"`
|
||||||
Timeouts Timeouts `yaml:"timeouts"`
|
Timeouts Timeouts `yaml:"timeouts"`
|
||||||
|
// Capabilities is the DEFAULT wire shape for this provider's models (D3.1);
|
||||||
|
// a per-Model block overrides it field-by-field. Nil = OpenAI-compat
|
||||||
|
// baseline (max_tokens + send temperature + off-by-omission reasoning).
|
||||||
|
Capabilities *CapabilitiesConfig `yaml:"capabilities"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// CapabilitiesConfig is the declarative per-model wire shape (D3.1). Empty
|
||||||
|
// fields inherit: a Model block layers over its Provider block, which layers
|
||||||
|
// over the kind baseline. Separate from Provider.Reasoning (subset|additive),
|
||||||
|
// which is a billing semantic — this is the wire body form.
|
||||||
|
type CapabilitiesConfig struct {
|
||||||
|
BudgetField string `yaml:"budget_field"` // max_tokens | max_completion_tokens
|
||||||
|
Temperature *TemperatureCap `yaml:"temperature"`
|
||||||
|
Reasoning *ReasoningCapCfg `yaml:"reasoning"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// TemperatureCap declares how temperature reaches the wire.
|
||||||
|
type TemperatureCap struct {
|
||||||
|
Mode string `yaml:"mode"` // send | omit | force
|
||||||
|
Value float64 `yaml:"value"` // used when mode == force
|
||||||
|
}
|
||||||
|
|
||||||
|
// ReasoningCapCfg declares how the neutral reasoning effort maps to the wire.
|
||||||
|
type ReasoningCapCfg struct {
|
||||||
|
Control string `yaml:"control"` // none | effort | extra_body_disable | mandatory
|
||||||
|
OffEffort string `yaml:"off_effort"` // effort: value sent when reasoning=off
|
||||||
|
OffExtraBody map[string]any `yaml:"off_extra_body"` // extra_body_disable: merged when reasoning=off
|
||||||
|
OnExtraBody map[string]any `yaml:"on_extra_body"` // extra_body_disable: merged when reasoning is on (D6.2)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Timeouts is the retry profile per provider (профиль per-провайдер из
|
// Timeouts is the retry profile per provider (профиль per-провайдер из
|
||||||
|
|
@ -73,6 +101,9 @@ type Model struct {
|
||||||
// (короче — кэш молча не создаётся); сборщик контекста Фазы 1 сверяется.
|
// (короче — кэш молча не создаётся); сборщик контекста Фазы 1 сверяется.
|
||||||
MinCachePrefixTokens int `yaml:"min_cache_prefix_tokens"`
|
MinCachePrefixTokens int `yaml:"min_cache_prefix_tokens"`
|
||||||
Note string `yaml:"note"`
|
Note string `yaml:"note"`
|
||||||
|
// Capabilities overrides this model's provider-default wire shape (D3.1),
|
||||||
|
// field-by-field (model wins). Nil = inherit the provider/kind baseline.
|
||||||
|
Capabilities *CapabilitiesConfig `yaml:"capabilities"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// Price mirrors ledger.ModelPrice in YAML form (USD per 1M tokens).
|
// Price mirrors ledger.ModelPrice in YAML form (USD per 1M tokens).
|
||||||
|
|
@ -140,6 +171,7 @@ func LoadModels(path string) (*Models, error) {
|
||||||
if p.Kind == "local" && p.Model == "" {
|
if p.Kind == "local" && p.Model == "" {
|
||||||
bad("provider %s: local kind requires model (its own tag)", name)
|
bad("provider %s: local kind requires model (its own tag)", name)
|
||||||
}
|
}
|
||||||
|
validateCapabilities(bad, "provider "+name, p.Capabilities)
|
||||||
}
|
}
|
||||||
for name, mod := range m.Models {
|
for name, mod := range m.Models {
|
||||||
prov, ok := m.Providers[mod.Provider]
|
prov, ok := m.Providers[mod.Provider]
|
||||||
|
|
@ -151,6 +183,7 @@ func LoadModels(path string) (*Models, error) {
|
||||||
if !isLocal && (mod.Price.InputPerM <= 0 || mod.Price.OutputPerM <= 0) {
|
if !isLocal && (mod.Price.InputPerM <= 0 || mod.Price.OutputPerM <= 0) {
|
||||||
bad("model %s: non-local models require non-zero input/output prices", name)
|
bad("model %s: non-local models require non-zero input/output prices", name)
|
||||||
}
|
}
|
||||||
|
validateCapabilities(bad, "model "+name, mod.Capabilities)
|
||||||
}
|
}
|
||||||
if len(problems) > 0 {
|
if len(problems) > 0 {
|
||||||
return nil, fmt.Errorf("config %s:\n - %s", path, strings.Join(problems, "\n - "))
|
return nil, fmt.Errorf("config %s:\n - %s", path, strings.Join(problems, "\n - "))
|
||||||
|
|
@ -185,6 +218,94 @@ func (p Provider) APIKey() string {
|
||||||
return os.Getenv(p.APIKeyEnv)
|
return os.Getenv(p.APIKeyEnv)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ResolveCapability computes the wire shape for a model (D3.1): the per-kind
|
||||||
|
// baseline, then the provider-default capabilities, then the per-model override
|
||||||
|
// (model wins, field-by-field). The runner uses the SAME method to fold the
|
||||||
|
// capability into the job snapshot, so what is hashed always matches what is
|
||||||
|
// sent — editing a capability invalidates checkpoints instead of false-hitting.
|
||||||
|
func (m *Models) ResolveCapability(modelName string) llm.Capability {
|
||||||
|
mod := m.Models[modelName]
|
||||||
|
prov := m.Providers[mod.Provider]
|
||||||
|
resolved := baselineCapability(prov.Kind)
|
||||||
|
applyCapConfig(&resolved, prov.Capabilities)
|
||||||
|
applyCapConfig(&resolved, mod.Capabilities)
|
||||||
|
return resolved
|
||||||
|
}
|
||||||
|
|
||||||
|
// baselineCapability is the per-kind default before any capabilities block: the
|
||||||
|
// OpenAI-compat wire (max_tokens + send temperature), with reasoning off by
|
||||||
|
// omission for cloud and by the ollama "none" switch for local.
|
||||||
|
func baselineCapability(kind string) llm.Capability {
|
||||||
|
c := llm.Capability{Budget: llm.BudgetMaxTokens, Temp: llm.TempSend}
|
||||||
|
if kind == "local" {
|
||||||
|
c.Reasoning = llm.ReasoningCap{Control: llm.ReasoningEffortField, OffEffort: "none"}
|
||||||
|
} else {
|
||||||
|
c.Reasoning = llm.ReasoningCap{Control: llm.ReasoningNone}
|
||||||
|
}
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
// applyCapConfig layers a capabilities block over a resolved capability. Each
|
||||||
|
// top-level field (budget / temperature / reasoning) inherits when the block
|
||||||
|
// leaves it empty, so a model overrides only what it declares.
|
||||||
|
func applyCapConfig(c *llm.Capability, cfg *CapabilitiesConfig) {
|
||||||
|
if cfg == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if cfg.BudgetField != "" {
|
||||||
|
c.Budget = llm.BudgetField(cfg.BudgetField)
|
||||||
|
}
|
||||||
|
if cfg.Temperature != nil {
|
||||||
|
c.Temp = llm.TempMode(cfg.Temperature.Mode)
|
||||||
|
c.TempValue = cfg.Temperature.Value
|
||||||
|
}
|
||||||
|
if cfg.Reasoning != nil {
|
||||||
|
c.Reasoning = llm.ReasoningCap{
|
||||||
|
Control: llm.ReasoningControl(cfg.Reasoning.Control),
|
||||||
|
OffEffort: cfg.Reasoning.OffEffort,
|
||||||
|
OffExtraBody: cfg.Reasoning.OffExtraBody,
|
||||||
|
OnExtraBody: cfg.Reasoning.OnExtraBody,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// validateCapabilities fail-fasts a capabilities block's enum fields (part of
|
||||||
|
// the models.yaml problem list, so a typo is caught at load, not as a 4xx).
|
||||||
|
func validateCapabilities(bad func(string, ...any), where string, c *CapabilitiesConfig) {
|
||||||
|
if c == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
switch c.BudgetField {
|
||||||
|
case "", "max_tokens", "max_completion_tokens":
|
||||||
|
default:
|
||||||
|
bad("%s: capabilities.budget_field must be max_tokens|max_completion_tokens, got %q", where, c.BudgetField)
|
||||||
|
}
|
||||||
|
if c.Temperature != nil {
|
||||||
|
switch c.Temperature.Mode {
|
||||||
|
case "send", "omit", "force":
|
||||||
|
default:
|
||||||
|
bad("%s: capabilities.temperature.mode must be send|omit|force, got %q", where, c.Temperature.Mode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if c.Reasoning != nil {
|
||||||
|
switch c.Reasoning.Control {
|
||||||
|
case "none", "effort", "extra_body_disable", "mandatory":
|
||||||
|
default:
|
||||||
|
bad("%s: capabilities.reasoning.control must be none|effort|extra_body_disable|mandatory, got %q", where, c.Reasoning.Control)
|
||||||
|
}
|
||||||
|
// The control must carry the companion field it consumes, else the
|
||||||
|
// disable silently no-ops at runtime (a typo'd off_extra_body key is
|
||||||
|
// dropped by non-strict yaml → nil → thinking stays ON). Catch it at
|
||||||
|
// load, not as a paid-but-empty completion.
|
||||||
|
if c.Reasoning.Control == "extra_body_disable" && len(c.Reasoning.OffExtraBody) == 0 {
|
||||||
|
bad("%s: capabilities.reasoning.control=extra_body_disable requires a non-empty off_extra_body (the disable switch)", where)
|
||||||
|
}
|
||||||
|
if c.Reasoning.Control == "effort" && c.Reasoning.OffEffort == "" {
|
||||||
|
bad("%s: capabilities.reasoning.control=effort requires off_effort (the value sent when reasoning=off)", where)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// CheckKeys verifies that every non-local model the pipeline can ACTUALLY call
|
// CheckKeys verifies that every non-local model the pipeline can ACTUALLY call
|
||||||
// has its provider API key resolvable in the environment — preflight fail-fast,
|
// has its provider API key resolvable in the environment — preflight fail-fast,
|
||||||
// чтобы пустой ключ не всплывал 401-м уже ПОСЛЕ reserve/списания слота.
|
// чтобы пустой ключ не всплывал 401-м уже ПОСЛЕ reserve/списания слота.
|
||||||
|
|
|
||||||
162
backend/internal/llm/capability.go
Normal file
162
backend/internal/llm/capability.go
Normal file
|
|
@ -0,0 +1,162 @@
|
||||||
|
package llm
|
||||||
|
|
||||||
|
// capability.go is the per-MODEL wire-shape layer (03-decisions-log D3.1).
|
||||||
|
// OpenAI-compatible providers share an ENDPOINT but not a BODY: Kimi accepts
|
||||||
|
// only temperature:1 (else 400), gpt-5-mini needs max_completion_tokens WITHOUT
|
||||||
|
// temperature (else 400), Gemini 3.1 Pro 400s if thinking is disabled. The
|
||||||
|
// Phase-0 wire — hardcoded max_tokens + always-send temperature — 400s on all
|
||||||
|
// three. Capability declares, per model (resolved from models.yaml: provider
|
||||||
|
// default merged with a per-model override, model wins), how to render
|
||||||
|
// budget / temperature / reasoning into the wire body.
|
||||||
|
//
|
||||||
|
// The resolver runs at openAIRequest marshal time so the neutral request stays
|
||||||
|
// vendor-free, and the resolved Capability is folded into the job snapshot
|
||||||
|
// (stageSnap) so editing a capability INVALIDATES checkpoints instead of
|
||||||
|
// silently false-hitting them — the same determinism class the snapshot payload
|
||||||
|
// closes (D5.2). Capability is the WIRE form; it is deliberately separate from
|
||||||
|
// Provider.Reasoning (subset|additive), which is a BILLING semantic feeding the
|
||||||
|
// ledger — a model can bill additively yet still disable thinking on the wire.
|
||||||
|
|
||||||
|
// BudgetField is the JSON key carrying the output-token cap.
|
||||||
|
type BudgetField string
|
||||||
|
|
||||||
|
const (
|
||||||
|
// BudgetMaxTokens is the classic OpenAI field (DeepSeek, GLM, xAI, Kimi,
|
||||||
|
// Gemini-compat, local).
|
||||||
|
BudgetMaxTokens BudgetField = "max_tokens"
|
||||||
|
// BudgetMaxCompletionTokens is the gpt-5 family field; sending max_tokens
|
||||||
|
// there is rejected.
|
||||||
|
BudgetMaxCompletionTokens BudgetField = "max_completion_tokens"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TempMode decides whether and how temperature reaches the wire.
|
||||||
|
type TempMode string
|
||||||
|
|
||||||
|
const (
|
||||||
|
// TempSend emits the request's temperature verbatim, INCLUDING an explicit
|
||||||
|
// 0 — a deterministic role (judge) must not silently sample at the provider
|
||||||
|
// default ~1.0 (the Phase-0 no-omitempty fix, now expressed per-model).
|
||||||
|
TempSend TempMode = "send"
|
||||||
|
// TempOmit never emits temperature (gpt-5 family: any temperature 400s).
|
||||||
|
TempOmit TempMode = "omit"
|
||||||
|
// TempForce always emits TempValue, ignoring the request (Kimi: only
|
||||||
|
// temperature:1 is accepted).
|
||||||
|
TempForce TempMode = "force"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ReasoningControl maps the neutral ReasoningEffort ("" | off | low | medium |
|
||||||
|
// high) onto the wire.
|
||||||
|
type ReasoningControl string
|
||||||
|
|
||||||
|
const (
|
||||||
|
// ReasoningNone is the OpenAI-compat baseline: low|medium|high go out as
|
||||||
|
// reasoning_effort; off and "" emit nothing (the provider's own default).
|
||||||
|
// Fits models where thinking is off by omission (DeepSeek / xAI-Grok on the
|
||||||
|
// Phase-1 draft/edit path).
|
||||||
|
ReasoningNone ReasoningControl = "none"
|
||||||
|
// ReasoningEffortField maps "off" to a fixed OffEffort value on the
|
||||||
|
// reasoning_effort field (gpt-5: "minimal"; local ollama: "none").
|
||||||
|
ReasoningEffortField ReasoningControl = "effort"
|
||||||
|
// ReasoningExtraBodyDisable disables thinking by MERGING OffExtraBody into
|
||||||
|
// the body (GLM {"thinking":{"type":"disabled"}}); for a think-ON escalation
|
||||||
|
// (D6.2) it merges OnExtraBody. effort "" leaves the provider default.
|
||||||
|
ReasoningExtraBodyDisable ReasoningControl = "extra_body_disable"
|
||||||
|
// ReasoningMandatory means thinking CANNOT be disabled — any disable attempt
|
||||||
|
// (reasoning_effort, thinking_budget:0) 400s (Gemini 3.1 Pro). The neutral
|
||||||
|
// "off" is swallowed: nothing reasoning-related is emitted, the model's
|
||||||
|
// forced thinking runs and MUST be budgeted as output ($12/M, D6.3).
|
||||||
|
ReasoningMandatory ReasoningControl = "mandatory"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ReasoningCap is the resolved reasoning wire-form for one model.
|
||||||
|
type ReasoningCap struct {
|
||||||
|
Control ReasoningControl
|
||||||
|
OffEffort string // ReasoningEffortField: value sent when effort=="off"
|
||||||
|
OffExtraBody map[string]any // ReasoningExtraBodyDisable: merged when effort=="off"
|
||||||
|
OnExtraBody map[string]any // ReasoningExtraBodyDisable: merged when thinking is on (D6.2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capability is the fully-resolved wire shape for one model. Its zero value is
|
||||||
|
// the OpenAI-compat baseline (see withDefaults), so a model with no
|
||||||
|
// capabilities block and the direct-construction path in tests both reproduce
|
||||||
|
// the Phase-0 wire.
|
||||||
|
type Capability struct {
|
||||||
|
Budget BudgetField
|
||||||
|
Temp TempMode
|
||||||
|
TempValue float64
|
||||||
|
Reasoning ReasoningCap
|
||||||
|
}
|
||||||
|
|
||||||
|
// withDefaults fills the OpenAI-compat baseline for zero fields: max_tokens +
|
||||||
|
// always-send temperature + off-by-omission reasoning.
|
||||||
|
func (c Capability) withDefaults() Capability {
|
||||||
|
if c.Budget == "" {
|
||||||
|
c.Budget = BudgetMaxTokens
|
||||||
|
}
|
||||||
|
if c.Temp == "" {
|
||||||
|
c.Temp = TempSend
|
||||||
|
}
|
||||||
|
if c.Reasoning.Control == "" {
|
||||||
|
c.Reasoning.Control = ReasoningNone
|
||||||
|
}
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
// applyToBody writes the budget / temperature / reasoning keys into the wire
|
||||||
|
// body map. effort is the request's neutral ReasoningEffort. Resolved standard
|
||||||
|
// keys are authoritative; config-supplied extra_body merges only into the gaps
|
||||||
|
// (see openAIRequest.MarshalJSON and mergeBody).
|
||||||
|
func (c Capability) applyToBody(m map[string]any, maxTokens int, temperature float64, effort string) {
|
||||||
|
c = c.withDefaults()
|
||||||
|
m[string(c.Budget)] = maxTokens
|
||||||
|
|
||||||
|
switch c.Temp {
|
||||||
|
case TempOmit:
|
||||||
|
// never emitted
|
||||||
|
case TempForce:
|
||||||
|
m["temperature"] = c.TempValue
|
||||||
|
default: // TempSend
|
||||||
|
m["temperature"] = temperature
|
||||||
|
}
|
||||||
|
|
||||||
|
switch c.Reasoning.Control {
|
||||||
|
case ReasoningMandatory:
|
||||||
|
// swallow everything: a disable attempt 400s, forced thinking runs
|
||||||
|
case ReasoningExtraBodyDisable:
|
||||||
|
// A disable-capability's whole purpose is to keep thinking OFF: for GLM
|
||||||
|
// the provider default IS thinking-on (the ×3-timeout blowup the switch
|
||||||
|
// exists to prevent), so an UNSPECIFIED effort ("") defaults to disabled
|
||||||
|
// exactly like "off". This preserves the Phase-0 unconditional-disable
|
||||||
|
// semantics (extra_body merged on every call); only an explicit
|
||||||
|
// low|medium|high opts into think-ON via OnExtraBody (D6.2). This is a
|
||||||
|
// deliberate per-control reading of "" — it differs from ReasoningNone,
|
||||||
|
// where "" means the provider default.
|
||||||
|
if effort == "" || effort == "off" {
|
||||||
|
mergeBody(m, c.Reasoning.OffExtraBody)
|
||||||
|
} else {
|
||||||
|
mergeBody(m, c.Reasoning.OnExtraBody)
|
||||||
|
}
|
||||||
|
case ReasoningEffortField:
|
||||||
|
if effort == "off" {
|
||||||
|
if c.Reasoning.OffEffort != "" {
|
||||||
|
m["reasoning_effort"] = c.Reasoning.OffEffort
|
||||||
|
}
|
||||||
|
} else if effort != "" {
|
||||||
|
m["reasoning_effort"] = effort
|
||||||
|
}
|
||||||
|
default: // ReasoningNone
|
||||||
|
if effort != "" && effort != "off" {
|
||||||
|
m["reasoning_effort"] = effort
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// mergeBody copies src into dst without overwriting an existing key: config
|
||||||
|
// tuning fills gaps, it never overrides a resolved standard field.
|
||||||
|
func mergeBody(dst, src map[string]any) {
|
||||||
|
for k, v := range src {
|
||||||
|
if _, exists := dst[k]; !exists {
|
||||||
|
dst[k] = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
210
backend/internal/llm/capability_test.go
Normal file
210
backend/internal/llm/capability_test.go
Normal file
|
|
@ -0,0 +1,210 @@
|
||||||
|
package llm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// capability_test.go asserts the exact wire body the D3.1 resolver renders for
|
||||||
|
// each real provider quirk (00-provider-quirks): the hardcoded Phase-0 body
|
||||||
|
// 400s on Kimi / gpt-5 / Gemini, so the body shape IS the contract.
|
||||||
|
|
||||||
|
func marshalBody(t *testing.T, r openAIRequest) map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
data, err := json.Marshal(r)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("marshal: %v", err)
|
||||||
|
}
|
||||||
|
var m map[string]any
|
||||||
|
if err := json.Unmarshal(data, &m); err != nil {
|
||||||
|
t.Fatalf("unmarshal: %v", err)
|
||||||
|
}
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustNum(t *testing.T, m map[string]any, k string, want float64) {
|
||||||
|
t.Helper()
|
||||||
|
v, ok := m[k]
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("key %q absent, want %v", k, want)
|
||||||
|
}
|
||||||
|
if f, ok := v.(float64); !ok || f != want {
|
||||||
|
t.Fatalf("key %q = %v, want %v", k, v, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustStr(t *testing.T, m map[string]any, k, want string) {
|
||||||
|
t.Helper()
|
||||||
|
if v, _ := m[k].(string); v != want {
|
||||||
|
t.Fatalf("key %q = %v, want %q", k, m[k], want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustAbsent(t *testing.T, m map[string]any, k string) {
|
||||||
|
t.Helper()
|
||||||
|
if _, ok := m[k]; ok {
|
||||||
|
t.Fatalf("key %q must be absent, got %v", k, m[k])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCapabilityWireBody(t *testing.T) {
|
||||||
|
base := openAIRequest{
|
||||||
|
model: "m",
|
||||||
|
messages: []openAIMessage{{Role: "user", Content: "u"}},
|
||||||
|
maxTokens: 4096,
|
||||||
|
temperature: 0.4,
|
||||||
|
}
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
cap Capability
|
||||||
|
effort string
|
||||||
|
extra map[string]any
|
||||||
|
check func(t *testing.T, m map[string]any)
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "zero cap reproduces the Phase-0 wire",
|
||||||
|
effort: "off",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustNum(t, m, "max_tokens", 4096)
|
||||||
|
mustNum(t, m, "temperature", 0.4)
|
||||||
|
mustAbsent(t, m, "max_completion_tokens")
|
||||||
|
mustAbsent(t, m, "reasoning_effort")
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "gpt-5: max_completion_tokens, omit temperature, off->minimal",
|
||||||
|
cap: Capability{Budget: BudgetMaxCompletionTokens, Temp: TempOmit, Reasoning: ReasoningCap{Control: ReasoningEffortField, OffEffort: "minimal"}},
|
||||||
|
effort: "off",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustNum(t, m, "max_completion_tokens", 4096)
|
||||||
|
mustAbsent(t, m, "max_tokens")
|
||||||
|
mustAbsent(t, m, "temperature")
|
||||||
|
mustStr(t, m, "reasoning_effort", "minimal")
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "gemini mandatory: swallow off, never disable, temperature sent",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningMandatory}},
|
||||||
|
effort: "off",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustNum(t, m, "max_tokens", 4096)
|
||||||
|
mustNum(t, m, "temperature", 0.4)
|
||||||
|
mustAbsent(t, m, "reasoning_effort")
|
||||||
|
mustAbsent(t, m, "thinking")
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "gemini mandatory swallows a NON-off effort too (escalation at high still 400-safe)",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningMandatory}},
|
||||||
|
effort: "high",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustAbsent(t, m, "reasoning_effort")
|
||||||
|
mustAbsent(t, m, "thinking")
|
||||||
|
mustNum(t, m, "temperature", 0.4)
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "GLM extra_body_disable: off merges thinking:disabled, no reasoning_effort",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningExtraBodyDisable, OffExtraBody: map[string]any{"thinking": map[string]any{"type": "disabled"}}}},
|
||||||
|
effort: "off",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustAbsent(t, m, "reasoning_effort")
|
||||||
|
th, ok := m["thinking"].(map[string]any)
|
||||||
|
if !ok || th["type"] != "disabled" {
|
||||||
|
t.Fatalf("thinking disable not merged: %v", m["thinking"])
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "GLM extra_body_disable: UNSET effort also disables (Phase-0 unconditional-disable preserved)",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningExtraBodyDisable, OffExtraBody: map[string]any{"thinking": map[string]any{"type": "disabled"}}}},
|
||||||
|
effort: "",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustAbsent(t, m, "reasoning_effort")
|
||||||
|
th, ok := m["thinking"].(map[string]any)
|
||||||
|
if !ok || th["type"] != "disabled" {
|
||||||
|
t.Fatalf("unset effort must still disable thinking (else GLM ×3 timeouts): %v", m["thinking"])
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "GLM think-ON: on merges on_extra_body (D6.2)",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningExtraBodyDisable, OnExtraBody: map[string]any{"thinking": map[string]any{"type": "enabled"}}}},
|
||||||
|
effort: "high",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
th, ok := m["thinking"].(map[string]any)
|
||||||
|
if !ok || th["type"] != "enabled" {
|
||||||
|
t.Fatalf("think-ON not merged: %v", m["thinking"])
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "kimi force temperature=1 regardless of the request",
|
||||||
|
cap: Capability{Temp: TempForce, TempValue: 1},
|
||||||
|
effort: "off",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustNum(t, m, "temperature", 1)
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "reasoning none passes low/medium/high through",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningNone}},
|
||||||
|
effort: "medium",
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustStr(t, m, "reasoning_effort", "medium")
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "extra fills gaps only, never overrides a resolved key",
|
||||||
|
effort: "off",
|
||||||
|
extra: map[string]any{"max_tokens": 999, "custom": "x"},
|
||||||
|
check: func(t *testing.T, m map[string]any) {
|
||||||
|
mustNum(t, m, "max_tokens", 4096) // resolved wins
|
||||||
|
mustStr(t, m, "custom", "x") // gap filled
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range tests {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
r := base
|
||||||
|
r.cap = tc.cap
|
||||||
|
r.reasoningEffort = tc.effort
|
||||||
|
r.extra = tc.extra
|
||||||
|
m := marshalBody(t, r)
|
||||||
|
if m["stream"] != false {
|
||||||
|
t.Fatalf("stream must serialize as false, got %v", m["stream"])
|
||||||
|
}
|
||||||
|
tc.check(t, m)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGLMMigrationWireIdentical proves the D3.1 migration of GLM's thinking
|
||||||
|
// disable from model.extra_body to capabilities.reasoning.off_extra_body leaves
|
||||||
|
// the WIRE body byte-identical — the snapshot changes (an intended one-time
|
||||||
|
// resnapshot) but not the request, so no live GLM call diverges (D5.2).
|
||||||
|
func TestGLMMigrationWireIdentical(t *testing.T) {
|
||||||
|
msgs := []openAIMessage{{Role: "system", Content: "s"}, {Role: "user", Content: "u"}}
|
||||||
|
legacy := openAIRequest{ // Phase-0 glm-5: default cap + extra_body disable switch
|
||||||
|
model: "glm-5", messages: msgs, maxTokens: 4096, temperature: 0.4, reasoningEffort: "off",
|
||||||
|
extra: map[string]any{"thinking": map[string]any{"type": "disabled"}},
|
||||||
|
}
|
||||||
|
migrated := openAIRequest{ // Phase-1 glm-5: disable via capabilities.reasoning.off_extra_body
|
||||||
|
model: "glm-5", messages: msgs, maxTokens: 4096, temperature: 0.4, reasoningEffort: "off",
|
||||||
|
cap: Capability{Reasoning: ReasoningCap{Control: ReasoningExtraBodyDisable, OffExtraBody: map[string]any{"thinking": map[string]any{"type": "disabled"}}}},
|
||||||
|
}
|
||||||
|
lb, err := json.Marshal(legacy)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mb, err := json.Marshal(migrated)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(lb, mb) {
|
||||||
|
t.Fatalf("GLM migration changed the wire body:\n legacy=%s\n migrated=%s", lb, mb)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -159,44 +159,43 @@ type openAIMessage struct {
|
||||||
Content string `json:"content"`
|
Content string `json:"content"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// openAIRequest is the neutral pre-wire request. Its body is assembled by
|
||||||
|
// MarshalJSON THROUGH the per-model Capability resolver (capability.go), not
|
||||||
|
// from fixed json tags: OpenAI-compatible providers share an endpoint but not a
|
||||||
|
// body — the budget key, whether temperature is emitted, and the reasoning
|
||||||
|
// switch all vary per model (D3.1). The Phase-0 fixed body (max_tokens +
|
||||||
|
// always-send temperature + reasoning_effort) 400s on Kimi / gpt-5 / Gemini;
|
||||||
|
// the resolver renders the correct shape and is itself folded into the job
|
||||||
|
// snapshot, so changing it invalidates checkpoints rather than false-hitting.
|
||||||
type openAIRequest struct {
|
type openAIRequest struct {
|
||||||
Model string `json:"model"`
|
model string
|
||||||
Messages []openAIMessage `json:"messages"`
|
messages []openAIMessage
|
||||||
MaxTokens int `json:"max_tokens"`
|
maxTokens int
|
||||||
// Temperature is ALWAYS sent (no omitempty): дропнутый explicit T=0
|
temperature float64
|
||||||
// молча превращал детерминированную стадию (судья C2) в сэмплирование
|
stream bool
|
||||||
// провайдерским дефолтом ~1.0.
|
// reasoningEffort is the NEUTRAL effort ("" | off | low | medium | high);
|
||||||
Temperature float64 `json:"temperature"`
|
// cap decides how — and whether — it reaches the wire.
|
||||||
Stream bool `json:"stream"`
|
reasoningEffort string
|
||||||
// Optional; omitempty keeps the plain body minimal.
|
responseFormat any
|
||||||
ReasoningEffort string `json:"reasoning_effort,omitempty"`
|
cap Capability
|
||||||
ResponseFormat any `json:"response_format,omitempty"`
|
// extra carries per-model provider-specific fields from models.yaml. Merged
|
||||||
// extra carries per-model provider-specific fields from models.yaml
|
// LAST and only into gaps — a resolved standard key always wins.
|
||||||
// (например GLM {"thinking":{"type":"disabled"}}). Merged into the JSON
|
|
||||||
// body at marshal time — the neutral request stays vendor-free.
|
|
||||||
extra map[string]any
|
extra map[string]any
|
||||||
}
|
}
|
||||||
|
|
||||||
// MarshalJSON merges extra into the standard body. Standard fields win on
|
// MarshalJSON assembles the wire body via the capability resolver. Resolved
|
||||||
// key collision (extra is config-supplied tuning, not an override channel).
|
// standard keys are authoritative; config-supplied extra_body fills only gaps.
|
||||||
func (r openAIRequest) MarshalJSON() ([]byte, error) {
|
func (r openAIRequest) MarshalJSON() ([]byte, error) {
|
||||||
type plain openAIRequest
|
m := map[string]any{
|
||||||
base, err := json.Marshal(plain(r))
|
"model": r.model,
|
||||||
if err != nil {
|
"messages": r.messages,
|
||||||
return nil, err
|
"stream": r.stream,
|
||||||
}
|
}
|
||||||
if len(r.extra) == 0 {
|
r.cap.applyToBody(m, r.maxTokens, r.temperature, r.reasoningEffort)
|
||||||
return base, nil
|
if r.responseFormat != nil {
|
||||||
}
|
m["response_format"] = r.responseFormat
|
||||||
var m map[string]any
|
|
||||||
if err := json.Unmarshal(base, &m); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
for k, v := range r.extra {
|
|
||||||
if _, exists := m[k]; !exists {
|
|
||||||
m[k] = v
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
mergeBody(m, r.extra)
|
||||||
return json.Marshal(m)
|
return json.Marshal(m)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -69,7 +69,10 @@ type LLMRequest struct {
|
||||||
Model string
|
Model string
|
||||||
Messages []Message
|
Messages []Message
|
||||||
MaxTokens int
|
MaxTokens int
|
||||||
Temperature float64 // adapters send it only when > 0
|
// Temperature: how (and whether) it reaches the wire is decided per-model
|
||||||
|
// by the Capability resolver (capability.go: send incl. explicit 0 | omit |
|
||||||
|
// force) — the adapter no longer gates on > 0.
|
||||||
|
Temperature float64
|
||||||
// ReasoningEffort: "" = provider default, "off" = explicitly disable
|
// ReasoningEffort: "" = provider default, "off" = explicitly disable
|
||||||
// thinking, "low"|"medium"|"high" = enable at that depth. Each adapter maps
|
// thinking, "low"|"medium"|"high" = enable at that depth. Each adapter maps
|
||||||
// this to its wire form; providers with non-standard switches (GLM
|
// this to its wire form; providers with non-standard switches (GLM
|
||||||
|
|
|
||||||
|
|
@ -37,6 +37,10 @@ type LocalConfig struct {
|
||||||
// return empty content. 0 = inherit the request's.
|
// return empty content. 0 = inherit the request's.
|
||||||
MaxTokens int
|
MaxTokens int
|
||||||
Profile RetryProfile
|
Profile RetryProfile
|
||||||
|
// Cap is the resolved wire shape (capability.go). If its reasoning control
|
||||||
|
// is unset, NewLocalClient defaults it to the ollama off-switch (map "off"
|
||||||
|
// onto reasoning_effort:"none").
|
||||||
|
Cap Capability
|
||||||
}
|
}
|
||||||
|
|
||||||
// NoProxyClient is an http.Client that bypasses any environment proxy.
|
// NoProxyClient is an http.Client that bypasses any environment proxy.
|
||||||
|
|
@ -50,15 +54,25 @@ type localClient struct {
|
||||||
model string
|
model string
|
||||||
temp float64
|
temp float64
|
||||||
maxTok int
|
maxTok int
|
||||||
|
cap Capability
|
||||||
}
|
}
|
||||||
|
|
||||||
// NewLocalClient builds the local-backend adapter.
|
// NewLocalClient builds the local-backend adapter.
|
||||||
func NewLocalClient(cfg LocalConfig, logger *slog.Logger) LLMClient {
|
func NewLocalClient(cfg LocalConfig, logger *slog.Logger) LLMClient {
|
||||||
|
cap := cfg.Cap
|
||||||
|
if cap.Reasoning.Control == "" {
|
||||||
|
// ollama's /v1 maps reasoning_effort onto Qwen3 thinking; "none" is its
|
||||||
|
// explicit off-switch (verified in vojo). Default it here so a zero-Cap
|
||||||
|
// caller keeps the off→"none" behavior; BuildClient passes the same via
|
||||||
|
// the resolver, so the snapshot's stageSnap capability matches the wire.
|
||||||
|
cap.Reasoning = ReasoningCap{Control: ReasoningEffortField, OffEffort: "none"}
|
||||||
|
}
|
||||||
return &localClient{
|
return &localClient{
|
||||||
http: newOpenAIClient("local", cfg.BaseURL, cfg.APIKey, cfg.Profile, nil, NoProxyClient(), logger),
|
http: newOpenAIClient("local", cfg.BaseURL, cfg.APIKey, cfg.Profile, nil, NoProxyClient(), logger),
|
||||||
model: cfg.Model,
|
model: cfg.Model,
|
||||||
temp: cfg.Temperature,
|
temp: cfg.Temperature,
|
||||||
maxTok: cfg.MaxTokens,
|
maxTok: cfg.MaxTokens,
|
||||||
|
cap: cap,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -71,20 +85,15 @@ func (c *localClient) Complete(ctx context.Context, req LLMRequest) (*LLMRespons
|
||||||
if c.temp > 0 {
|
if c.temp > 0 {
|
||||||
temp = c.temp
|
temp = c.temp
|
||||||
}
|
}
|
||||||
effort := req.ReasoningEffort
|
|
||||||
if effort == "off" {
|
|
||||||
// ollama's /v1 maps reasoning_effort onto Qwen3 thinking; "none" is its
|
|
||||||
// explicit off-switch (verified in vojo).
|
|
||||||
effort = "none"
|
|
||||||
}
|
|
||||||
resp, err := c.http.complete(ctx, openAIRequest{
|
resp, err := c.http.complete(ctx, openAIRequest{
|
||||||
Model: c.model,
|
model: c.model,
|
||||||
Messages: toOpenAIMessages(req.Messages),
|
messages: toOpenAIMessages(req.Messages),
|
||||||
MaxTokens: maxTok,
|
maxTokens: maxTok,
|
||||||
Temperature: temp,
|
temperature: temp,
|
||||||
Stream: false,
|
stream: false,
|
||||||
ReasoningEffort: effort,
|
reasoningEffort: req.ReasoningEffort, // neutral; cap maps off→"none" for ollama
|
||||||
ResponseFormat: jsonResponseFormat(req.JSONOnly),
|
responseFormat: jsonResponseFormat(req.JSONOnly),
|
||||||
|
cap: c.cap,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
|
|
|
||||||
|
|
@ -35,9 +35,13 @@ type OpenAICompatConfig struct {
|
||||||
Profile RetryProfile
|
Profile RetryProfile
|
||||||
Headers map[string]string // static extra headers, may be nil
|
Headers map[string]string // static extra headers, may be nil
|
||||||
Reasoning ReasoningSemantics
|
Reasoning ReasoningSemantics
|
||||||
// ExtraBody is merged into every request JSON for this provider's models
|
// Cap is the resolved per-model wire shape (capability.go). Zero value =
|
||||||
// (e.g. GLM {"thinking":{"type":"disabled"}}). Per-model extras from
|
// OpenAI-compat baseline (max_tokens + send temperature + off-by-omission).
|
||||||
// models.yaml are passed the same way when the client is built per-model.
|
Cap Capability
|
||||||
|
// ExtraBody is merged into every request JSON for this provider's models.
|
||||||
|
// Per-model extras from models.yaml are passed the same way when the client
|
||||||
|
// is built per-model. Provider-specific reasoning switches now live in Cap
|
||||||
|
// (capabilities.reasoning), not here; ExtraBody remains for other tuning.
|
||||||
ExtraBody map[string]any
|
ExtraBody map[string]any
|
||||||
// HTTPClient overrides the default client (nil = default). The local
|
// HTTPClient overrides the default client (nil = default). The local
|
||||||
// provider uses this for the no-proxy transport.
|
// provider uses this for the no-proxy transport.
|
||||||
|
|
@ -47,6 +51,7 @@ type OpenAICompatConfig struct {
|
||||||
type openAICompatClient struct {
|
type openAICompatClient struct {
|
||||||
http *openAIClient
|
http *openAIClient
|
||||||
reasoning ReasoningSemantics
|
reasoning ReasoningSemantics
|
||||||
|
cap Capability
|
||||||
extra map[string]any
|
extra map[string]any
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -59,25 +64,21 @@ func NewOpenAICompatClient(cfg OpenAICompatConfig, logger *slog.Logger) LLMClien
|
||||||
return &openAICompatClient{
|
return &openAICompatClient{
|
||||||
http: newOpenAIClient(cfg.Name, cfg.BaseURL, cfg.APIKey, cfg.Profile, cfg.Headers, cfg.HTTPClient, logger),
|
http: newOpenAIClient(cfg.Name, cfg.BaseURL, cfg.APIKey, cfg.Profile, cfg.Headers, cfg.HTTPClient, logger),
|
||||||
reasoning: sem,
|
reasoning: sem,
|
||||||
|
cap: cfg.Cap,
|
||||||
extra: cfg.ExtraBody,
|
extra: cfg.ExtraBody,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (c *openAICompatClient) Complete(ctx context.Context, req LLMRequest) (*LLMResponse, error) {
|
func (c *openAICompatClient) Complete(ctx context.Context, req LLMRequest) (*LLMResponse, error) {
|
||||||
effort := req.ReasoningEffort
|
|
||||||
if effort == "off" {
|
|
||||||
// "off" is our neutral value; OpenAI-compat providers with a
|
|
||||||
// non-standard disable switch get it via ExtraBody instead.
|
|
||||||
effort = ""
|
|
||||||
}
|
|
||||||
resp, err := c.http.complete(ctx, openAIRequest{
|
resp, err := c.http.complete(ctx, openAIRequest{
|
||||||
Model: req.Model,
|
model: req.Model,
|
||||||
Messages: toOpenAIMessages(req.Messages),
|
messages: toOpenAIMessages(req.Messages),
|
||||||
MaxTokens: req.MaxTokens,
|
maxTokens: req.MaxTokens,
|
||||||
Temperature: req.Temperature,
|
temperature: req.Temperature,
|
||||||
Stream: false,
|
stream: false,
|
||||||
ReasoningEffort: effort,
|
reasoningEffort: req.ReasoningEffort, // neutral; cap maps it to the wire
|
||||||
ResponseFormat: jsonResponseFormat(req.JSONOnly),
|
responseFormat: jsonResponseFormat(req.JSONOnly),
|
||||||
|
cap: c.cap,
|
||||||
extra: c.extra,
|
extra: c.extra,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|
|
||||||
|
|
@ -27,6 +27,7 @@ func BuildClient(models *config.Models, modelName string, logger *slog.Logger) (
|
||||||
APIKey: prov.APIKey(),
|
APIKey: prov.APIKey(),
|
||||||
Profile: prov.Timeouts.Profile(),
|
Profile: prov.Timeouts.Profile(),
|
||||||
Reasoning: llm.ReasoningSemantics(prov.Reasoning),
|
Reasoning: llm.ReasoningSemantics(prov.Reasoning),
|
||||||
|
Cap: models.ResolveCapability(modelName),
|
||||||
ExtraBody: mod.ExtraBody,
|
ExtraBody: mod.ExtraBody,
|
||||||
}, logger), nil
|
}, logger), nil
|
||||||
case "anthropic":
|
case "anthropic":
|
||||||
|
|
@ -44,6 +45,7 @@ func BuildClient(models *config.Models, modelName string, logger *slog.Logger) (
|
||||||
Temperature: prov.Temperature,
|
Temperature: prov.Temperature,
|
||||||
MaxTokens: prov.MaxTokens,
|
MaxTokens: prov.MaxTokens,
|
||||||
Profile: prov.Timeouts.Profile(),
|
Profile: prov.Timeouts.Profile(),
|
||||||
|
Cap: models.ResolveCapability(modelName),
|
||||||
}, logger), nil
|
}, logger), nil
|
||||||
default:
|
default:
|
||||||
return nil, fmt.Errorf("pipeline: provider %q has unknown kind %q", mod.Provider, prov.Kind)
|
return nil, fmt.Errorf("pipeline: provider %q has unknown kind %q", mod.Provider, prov.Kind)
|
||||||
|
|
|
||||||
|
|
@ -116,10 +116,36 @@ func (r *Runner) client(model string) (llm.LLMClient, error) {
|
||||||
return c, nil
|
return c, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// contextSnap freezes the context-assembly knobs (§3.8) inside the snapshot.
|
||||||
|
// Fixed field order — it is part of a content hash.
|
||||||
|
type contextSnap struct {
|
||||||
|
GlossaryInjection string `json:"glossary_injection"`
|
||||||
|
GlossaryTokenBudget int `json:"glossary_token_budget"`
|
||||||
|
STMDepth int `json:"stm_depth"`
|
||||||
|
OverlapTokens int `json:"overlap_tokens"`
|
||||||
|
CacheTTL string `json:"cache_ttl"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// memoryVersion is the content-hash of the deterministically materialized
|
||||||
|
// injected memory (approved glossary + summaries + series-bible), the memory
|
||||||
|
// component of the snapshot (D5.2/D8). Until the memory bank v2 migration lands
|
||||||
|
// the materialization is empty, so the hash is a stable constant; once memory
|
||||||
|
// enters msgs this changes and the resnapshot-gate fires loudly instead of a
|
||||||
|
// stale checkpoint re-paying a diverged translation. STM is excluded (rebuilt
|
||||||
|
// from checkpoints, §3.2).
|
||||||
|
func (r *Runner) memoryVersion() string {
|
||||||
|
h := sha256.New()
|
||||||
|
// Domain tag + empty materialization; the store query lands with the v2
|
||||||
|
// memory schema and feeds ORDER BY-stable rows here.
|
||||||
|
h.Write([]byte("tm-memory-v1\x00"))
|
||||||
|
return hex.EncodeToString(h.Sum(nil))
|
||||||
|
}
|
||||||
|
|
||||||
// snapshotID materializes the job-context snapshot (§3.2): brief_hash, chunker
|
// snapshotID materializes the job-context snapshot (§3.2): brief_hash, chunker
|
||||||
// version and the full stage plan (model, prompt version + CONTENT hash,
|
// version, context-assembly knobs, the memory version and the full stage plan
|
||||||
// sampling). Payload is rendered with a fixed field order — the id is a
|
// (model, prompt version + CONTENT hash, sampling, resolved capability).
|
||||||
// content hash, so identical context re-upserts idempotently.
|
// Payload is rendered with a fixed field order — the id is a content hash, so
|
||||||
|
// identical context re-upserts idempotently.
|
||||||
func (r *Runner) snapshotID() (id, payload string, err error) {
|
func (r *Runner) snapshotID() (id, payload string, err error) {
|
||||||
type stageSnap struct {
|
type stageSnap struct {
|
||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
|
|
@ -141,6 +167,12 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
|
||||||
// local-пути (находка внешнего ревью F2).
|
// local-пути (находка внешнего ревью F2).
|
||||||
ProviderTemp float64 `json:"provider_temp,omitempty"`
|
ProviderTemp float64 `json:"provider_temp,omitempty"`
|
||||||
ProviderMaxTok int `json:"provider_max_tok,omitempty"`
|
ProviderMaxTok int `json:"provider_max_tok,omitempty"`
|
||||||
|
// Capability — резолвнутая wire-форма модели (D3.1): budget-ключ,
|
||||||
|
// temperature-режим, reasoning-контроль. Меняет ТЕЛО запроса (max_tokens
|
||||||
|
// vs max_completion_tokens, отправлять ли temperature, thinking-выключа-
|
||||||
|
// тель), поэтому обязана входить в снапшот — иначе правка каппы молча
|
||||||
|
// false-хитит чекпоинты (тот же класс D5.2, что и payload ниже).
|
||||||
|
Capability json.RawMessage `json:"capability,omitempty"`
|
||||||
}
|
}
|
||||||
snap := struct {
|
snap := struct {
|
||||||
BriefHash string `json:"brief_hash"`
|
BriefHash string `json:"brief_hash"`
|
||||||
|
|
@ -150,9 +182,21 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
|
||||||
// Defaults влияют на maxTokens, а тот входит в request-hash: без них
|
// Defaults влияют на maxTokens, а тот входит в request-hash: без них
|
||||||
// правка max_output_ratio молча инвалидировала бы все чекпоинты в
|
// правка max_output_ratio молча инвалидировала бы все чекпоинты в
|
||||||
// обход snapshot-гейта (находка ревью).
|
// обход snapshot-гейта (находка ревью).
|
||||||
MaxOutputRatio float64 `json:"max_output_ratio"`
|
MaxOutputRatio float64 `json:"max_output_ratio"`
|
||||||
MinMaxTokens int `json:"min_max_tokens"`
|
MinMaxTokens int `json:"min_max_tokens"`
|
||||||
Stages []stageSnap `json:"stages"`
|
// ContextAssembly — ручки сборки контекста (глоссарий/STM/overlap/TTL,
|
||||||
|
// §3.8). Их правка меняет ВХОД модели, когда память войдёт в msgs; сворачи-
|
||||||
|
// ваем ДО инъекции, чтобы смена budget'а инъекции не промахнулась мимо
|
||||||
|
// resnapshot-гейта (D5.2). Фикс-порядок полей — это content-hash.
|
||||||
|
ContextAssembly contextSnap `json:"context_assembly"`
|
||||||
|
// MemoryVersion — content-hash детерминированно материализованной инъекти-
|
||||||
|
// руемой памяти (approved-глоссарий+резюме+series-bible). Не bump-счётчик
|
||||||
|
// (D8): забыть пересчитать нельзя, а забытый bump переоткрыл бы класс тихой
|
||||||
|
// переоплаты D5.2. До миграции банка памяти (v2) материализация пуста —
|
||||||
|
// хэш стабилен; когда память войдёт в msgs, поле сменится и resnapshot-
|
||||||
|
// гейт сработает громко. STM сюда НЕ входит (пересобирается из чекпоинтов).
|
||||||
|
MemoryVersion string `json:"memory_version"`
|
||||||
|
Stages []stageSnap `json:"stages"`
|
||||||
}{
|
}{
|
||||||
BriefHash: r.Book.BriefHash(),
|
BriefHash: r.Book.BriefHash(),
|
||||||
ChunkerVersion: chunkerVersion,
|
ChunkerVersion: chunkerVersion,
|
||||||
|
|
@ -160,6 +204,14 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
|
||||||
PipelineCore: r.Pipeline.Core,
|
PipelineCore: r.Pipeline.Core,
|
||||||
MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio,
|
MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio,
|
||||||
MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens,
|
MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens,
|
||||||
|
ContextAssembly: contextSnap{
|
||||||
|
GlossaryInjection: r.Pipeline.Context.GlossaryInjection,
|
||||||
|
GlossaryTokenBudget: r.Pipeline.Context.GlossaryTokenBudget,
|
||||||
|
STMDepth: r.Pipeline.Context.STMDepth,
|
||||||
|
OverlapTokens: r.Pipeline.Context.OverlapTokens,
|
||||||
|
CacheTTL: r.Pipeline.Context.CacheTTL,
|
||||||
|
},
|
||||||
|
MemoryVersion: r.memoryVersion(),
|
||||||
}
|
}
|
||||||
for _, st := range r.Pipeline.Stages {
|
for _, st := range r.Pipeline.Stages {
|
||||||
ss := stageSnap{
|
ss := stageSnap{
|
||||||
|
|
@ -177,6 +229,14 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
|
||||||
}
|
}
|
||||||
ss.ModelExtra = raw
|
ss.ModelExtra = raw
|
||||||
}
|
}
|
||||||
|
// Резолвнутая каппа — тем же ResolveCapability, что и у клиента, поэтому
|
||||||
|
// хэшируемое всегда совпадает с отправляемым. json.Marshal сортирует
|
||||||
|
// ключи карт (off_extra_body) → детерминированный рендер.
|
||||||
|
capRaw, cerr := json.Marshal(r.Models.ResolveCapability(st.Model))
|
||||||
|
if cerr != nil {
|
||||||
|
return "", "", cerr
|
||||||
|
}
|
||||||
|
ss.Capability = capRaw
|
||||||
snap.Stages = append(snap.Stages, ss)
|
snap.Stages = append(snap.Stages, ss)
|
||||||
}
|
}
|
||||||
data, err := json.Marshal(snap)
|
data, err := json.Marshal(snap)
|
||||||
|
|
|
||||||
|
|
@ -225,6 +225,60 @@ func TestRunnerSnapshotPinning(t *testing.T) {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// D5.2/B: правка ТОЛЬКО каппы модели (wire-форма) обязана сдвинуть snapshot и
|
||||||
|
// уронить resume громко — иначе изменённое тело запроса подалось бы из старого
|
||||||
|
// чекпоинта (тихий расходящийся ре-пэй). Гейт на то, что stageSnap реально
|
||||||
|
// сворачивает резолвнутую Capability: без `ss.Capability` этот тест бы прошёл
|
||||||
|
// resume молча (находка селфревью вехи 1).
|
||||||
|
func TestRunnerSnapshotPinsCapability(t *testing.T) {
|
||||||
|
var calls atomic.Int32
|
||||||
|
srv := newFakeProvider(t, &calls)
|
||||||
|
defer srv.Close()
|
||||||
|
bookPath := setupProject(t, srv.URL)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
r1, err := NewRunner(bookPath, obs.NewLogger())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := r1.TranslateOneChunk(ctx); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
r1.Close()
|
||||||
|
if calls.Load() != 2 {
|
||||||
|
t.Fatalf("run1 calls = %d", calls.Load())
|
||||||
|
}
|
||||||
|
|
||||||
|
// Меняем ТОЛЬКО каппу fake-model (temperature force) — промпт/сэмплинг/
|
||||||
|
// провайдер/цены те же. Резолвнутая Capability меняется → snapshotID обязан
|
||||||
|
// сдвинуться → resume падает на snapshot-pinning, а не подаёт чекпоинт.
|
||||||
|
modelsPath := filepath.Join(filepath.Dir(bookPath), "models.yaml")
|
||||||
|
raw, err := os.ReadFile(modelsPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
patched := strings.Replace(string(raw),
|
||||||
|
"output_per_m: 2.0 }",
|
||||||
|
"output_per_m: 2.0 }\n capabilities: { temperature: { mode: force, value: 0.9 } }", 1)
|
||||||
|
if patched == string(raw) {
|
||||||
|
t.Fatal("failed to inject capability into models.yaml")
|
||||||
|
}
|
||||||
|
writeFile(t, modelsPath, patched)
|
||||||
|
|
||||||
|
r2, err := NewRunner(bookPath, obs.NewLogger())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err = r2.TranslateOneChunk(ctx)
|
||||||
|
r2.Close()
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "resnapshot") {
|
||||||
|
t.Fatalf("a capability edit must fail loud mentioning --resnapshot (snapshot must fold the resolved capability), got: %v", err)
|
||||||
|
}
|
||||||
|
if calls.Load() != 2 {
|
||||||
|
t.Fatalf("denied resume must not call the provider, calls=%d", calls.Load())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Платный 2xx с нулевым usage не должен селтлиться в $0 (иначе потолок слепнет):
|
// Платный 2xx с нулевым usage не должен селтлиться в $0 (иначе потолок слепнет):
|
||||||
// берётся консервативная оценка резерва.
|
// берётся консервативная оценка резерва.
|
||||||
func TestRunnerZeroUsagePaidSettlesEstimate(t *testing.T) {
|
func TestRunnerZeroUsagePaidSettlesEstimate(t *testing.T) {
|
||||||
|
|
|
||||||
|
|
@ -239,6 +239,24 @@ keep-alive (Ф1–2), инъекция глоссария (`selective`), пор
|
||||||
> - **Q4 (D11): budget_usd — конфиг-потолок, не гейт приёмки** — ДА. Но ⚠ **оба cost-числа устарели, не только премиум:** дефолт-редактор grok-4.3 ($2.50/M out ≈ 10× draft) → стандарт-микс ранобэ ~$0.73 (>$0.6). Приёмка по стоимости = escalation уважает конфиг-потолок + **точность телеметрии** (ledger корректно биллит grok reasoning=0 / gemini reasoning-as-output), мягкий sanity ~$1/ранобэ, не жёсткий $0.6/$5. Точные числа — полигон пересчитывает на стеке draft-DeepSeek+editor-grok+премиум-Gemini.
|
> - **Q4 (D11): budget_usd — конфиг-потолок, не гейт приёмки** — ДА. Но ⚠ **оба cost-числа устарели, не только премиум:** дефолт-редактор grok-4.3 ($2.50/M out ≈ 10× draft) → стандарт-микс ранобэ ~$0.73 (>$0.6). Приёмка по стоимости = escalation уважает конфиг-потолок + **точность телеметрии** (ledger корректно биллит grok reasoning=0 / gemini reasoning-as-output), мягкий sanity ~$1/ранобэ, не жёсткий $0.6/$5. Точные числа — полигон пересчитывает на стеке draft-DeepSeek+editor-grok+премиум-Gemini.
|
||||||
> **Ре-флаг:** grok = дефолт-редактор → каждый чанк через xAI. Продакшн-редактор — **чистый xAI-аккаунт без data-sharing** (промо-аккаунт только eval/NSFW). Концентрация 3 ролей на xAI — риск доступности (Р10), gemini-премиум как замена под рукой.
|
> **Ре-флаг:** grok = дефолт-редактор → каждый чанк через xAI. Продакшн-редактор — **чистый xAI-аккаунт без data-sharing** (промо-аккаунт только eval/NSFW). Концентрация 3 ролей на xAI — риск доступности (Р10), gemini-премиум как замена под рукой.
|
||||||
|
|
||||||
|
### 2026-07-04 — Сессия 3: Веха 1 Фазы 1 — capability-слой (C) + snapshotID под память (B)
|
||||||
|
|
||||||
|
Ратифицированный порядок стартует с twin-фундамента (runner-loop+snapshotID, capability-слой вместе). Веха 1 = **C + B** (жёстко связаны через `stageSnap`); runner chapter/chunk-loop + A (disposition) — Веха 2. Всё зелёное: `go build/vet/test -race`, реальные конфиги валидируются через `tmctl report` ($0).
|
||||||
|
|
||||||
|
**C — capability-матрица (D3.1).** Декларативные `capabilities:` per-модель в `models.yaml`: `budget_field` (max_tokens|max_completion_tokens), `temperature` (send|omit|force+value), `reasoning` (control none|effort|extra_body_disable|mandatory + off_effort/off_extra_body/on_extra_body). Резолвер (`Capability.applyToBody`) строит тело в `openAIRequest.MarshalJSON`; `Models.ResolveCapability` мержит provider-дефолт ← model-оверрайд (модель побеждает по-полю). Fail-fast валидация (enum + companion-поле). **Нулевая каппа = Phase-0 wire** (обратная совместимость — все текущие модели без изменений на проводе). GLM `thinking:disabled` перенесён `extra_body` → `capabilities.reasoning.off_extra_body` (тело **байт-в-байт то же**, тест `TestGLMMigrationWireIdentical`; сдвигается только snapshot — осознанный `--resnapshot`). Отдельно от `Provider.Reasoning=subset|additive` (то — биллинг). Юнит-тесты покрывают все квирки: gpt-5 (max_completion_tokens+omit+minimal), Gemini mandatory-swallow, Kimi force=1, GLM off/on.
|
||||||
|
|
||||||
|
**B — snapshotID под память (D5.2/D8).** `snapshotID` сворачивает: (а) суб-хэш context-assembly-ручек (`glossary_injection`/budget/`stm_depth`/`overlap`/`cache_ttl`), (б) `memoryVersion` — **content-hash** материализованной инъектируемой памяти (не bump-счётчик, D8; пока пустая материализация до банка памяти v2), (в) резолвнутую capability per-стадию в `stageSnap`. STM исключён (пересобирается из чекпоинтов). Цепочка `capability → stageSnap → snapID → request_hash` подтверждена (mutation-verified: без фолда правка каппы молча подаёт старый чекпоинт).
|
||||||
|
|
||||||
|
**Заведён стек (частично).** Провайдер `xai` + `grok-4.3` (цена $1.25/$2.50 факт. 04.07). **`deepseek-v4-pro` ПОДТВЕРЖДЁН живьём** (`GET /models`: `[deepseek-v4-flash, deepseek-v4-pro]`; `deepseek-chat` из списка ушёл — депрекейт 24.07). Черновик = `deepseek-v4-flash`, голова эскалации A = `deepseek-v4-pro`. ⚠ Цену `v4-pro` `/models` не отдаёт — фактчекнуть перед wiring. **Отложено в след. конфиг-шаг** (нужны фактчек цен gpt-5-mini/nano + подтверждение xAI прод-ключа): переключение edit-стадии `glm-5`→`grok-4.3`, заводка gemini/openai + gemini-3.1-pro-preview/gpt-5-*/glm-5.1 + цепочки эскалации. Веха 1 даёт МЕХАНИЗМ (юнит-тесты по всем квиркам) + живой путь; полный стек — когда цены подтверждены.
|
||||||
|
|
||||||
|
**Селфревью (адверсариальный воркфлоу, 4 измерения → верификация каждой находки).** Детерминизм — **0 дефектов** (инвариант держит, проверено эмпирически). 5 находок исправлено: (1) *[регресс, мой]* GLM-disable стал условным на `reasoning=="off"` — пустой (незаданный) effort молча включал бы thinking (×3 таймауты + reasoning-биллинг); фикс: `extra_body_disable` трактует `""` как `"off"` (сохранён Phase-0 unconditional-disable), только явный low/med/high → think-ON; (2) валидация не требовала companion-поля (typo `off_extra_body` → тихий no-op) → требуем `off_extra_body`/`off_effort`; (3) нет теста, что каппа входит в `snapshotID` (центральный инвариант B) → e2e-гейт `TestRunnerSnapshotPinsCapability` (mutation-verified); (4) mandatory-swallow тестился только на `off` → строка на non-off; (5) валидация не тестилась на provider-пути/temp-mode/reasoning-control → добавлено.
|
||||||
|
|
||||||
|
**Техдолг вперёд (не блокирует Веху 2):** capabilities эскалационных моделей НЕ в snapshot (только стадии) — закрыть при заводке цикла эскалации (тот же класс D5.2, но пути исполнения ещё нет); `memoryVersion` — заглушка до банка памяти v2 (при инъекции памяти заполнить запрос); Kimi `max_tokens≥16000` не выразим в capability-схеме (при wiring эскалации).
|
||||||
|
|
||||||
|
**Гигиена xAI (напоминание):** в `models.yaml` у провайдера `xai` — комментарий-контракт «прод-clean без data-sharing». `XAI_API_KEY` в `.env` подтвердить как прод-аккаунт перед первым боевым grok-вызовом (в Вехе 1 grok в стадии не зовётся).
|
||||||
|
|
||||||
|
**Веха 2 (следующая):** runner chapter/chunk-loop + `chunk_status`/disposition (A, D2) — `flag_reason`-константы, `classify(text,finish)` (refusal/echo до length), ось attempt, exit-коды 0/2/1, фиксы F4 (усечённый length → flagged) + edit-`max_tokens` от длины черновика.
|
||||||
|
|
||||||
## Полигон
|
## Полигон
|
||||||
(секция параллельной сессии — записи добавлять сюда)
|
(секция параллельной сессии — записи добавлять сюда)
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue