Close external-review findings: fold classifier version and local backend tag into the snapshot, catch nested echo-mine forms, restrict escalation to the translator role

This commit is contained in:
Claude (backend session) 2026-07-05 01:22:40 +03:00
parent 3a5b966c16
commit 0b31222b5c
9 changed files with 237 additions and 34 deletions

View file

@ -81,6 +81,15 @@ models:
modelExtra: "\n extra_body: { thinking: { type: disabled } }",
wantErr: true,
},
{
// The NESTED form: chat_template_kwargs.thinking:false is DeepSeek's
// documented V3.1+ disable — the most likely real arming form. A flat
// top-level scan misses it; the recursive scan must catch it (ext-review).
name: "armed via nested chat_template_kwargs.thinking (DeepSeek V3.1+)",
echoProne: true,
modelExtra: "\n extra_body: { chat_template_kwargs: { thinking: false } }",
wantErr: true,
},
{
// reasoning_effort at off also pushes thinking off the default.
name: "armed via capabilities.reasoning.effort",

View file

@ -270,22 +270,50 @@ func (m *Models) ResolveCapability(modelName string) llm.Capability {
return resolved
}
// thinkingControlExtraKeys are model-level extra_body keys whose sole purpose is
// to toggle a provider's thinking on the wire (GLM/DeepSeek {"thinking":…}, Qwen
// {"enable_thinking":…}, a raw {"reasoning_effort":…}). On an echo-prone provider
// their mere PRESENCE is the mine, whatever the value: thinking must stay at the
// provider default, and these keys exist only to move it off that default.
var thinkingControlExtraKeys = []string{"thinking", "enable_thinking", "reasoning_effort"}
// thinkingControlExtraKeys are keys whose purpose is to toggle a provider's thinking
// on the wire — top-level (GLM/DeepSeek {"thinking":…}, Qwen {"enable_thinking":…}, a
// raw {"reasoning_effort":…}) OR nested (DeepSeek-V3.1+ disables via
// chat_template_kwargs.thinking:false). On an echo-prone provider their mere PRESENCE,
// at any depth, is the mine, whatever the value: thinking must stay at the provider
// default, and these keys exist only to move it off that default.
var thinkingControlExtraKeys = map[string]bool{
"thinking": true, "enable_thinking": true, "reasoning_effort": true, "reasoning": true,
}
// findThinkingControlKey RECURSIVELY scans an extra_body value for any thinking-control
// key, returning its dotted path (e.g. "chat_template_kwargs.thinking") or "". The
// recursion is what catches nested disable forms a flat top-level check would miss
// (external-review finding — chat_template_kwargs.thinking is DeepSeek's documented
// V3.1+ disable, the most likely real arming form).
func findThinkingControlKey(v any, path string) string {
m, ok := v.(map[string]any)
if !ok {
return ""
}
for k, sub := range m {
here := k
if path != "" {
here = path + "." + k
}
if thinkingControlExtraKeys[k] {
return here
}
if found := findThinkingControlKey(sub, here); found != "" {
return found
}
}
return ""
}
// echoMineViolation reports, for one model, whether its RESOLVED wire shape would
// suppress thinking at reasoning=off on an echo-prone provider — the DeepSeek echo
// mine (§2). It returns a human cause string, or "" when safe. Both injection
// surfaces are covered: the capabilities.reasoning switch (extra_body_disable
// merges a thinking-disable at off; effort sends reasoning_effort at off — both
// push thinking off the default) AND a thinking-control key smuggled through the
// model's top-level extra_body (merged into the wire body by openAIRequest). A
// non-echo-prone provider (GLM: its editor input is a Russian draft, nothing CJK
// to echo) is never gated, so its {thinking:{type:disabled}} stays legal.
// mine (§2). It returns a human cause string, or "" when safe. It covers both injection
// surfaces: the capabilities.reasoning switch (extra_body_disable merges a
// thinking-disable at off; effort sends reasoning_effort at off — both push thinking
// off the default) AND a thinking-control key anywhere in the model's extra_body,
// TOP-LEVEL OR NESTED (merged into the wire body by openAIRequest). A non-echo-prone
// provider (GLM: its editor input is a Russian draft, nothing CJK to echo) is never
// gated, so its {thinking:{type:disabled}} stays legal.
func (m *Models) echoMineViolation(name string) string {
mod := m.Models[name]
prov, ok := m.Providers[mod.Provider]
@ -298,10 +326,8 @@ func (m *Models) echoMineViolation(name string) string {
case llm.ReasoningEffortField:
return "capabilities.reasoning.control=effort would send reasoning_effort at reasoning=off, suppressing thinking"
}
for _, k := range thinkingControlExtraKeys {
if _, present := mod.ExtraBody[k]; present {
return fmt.Sprintf("extra_body carries the thinking-control key %q", k)
}
if path := findThinkingControlKey(map[string]any(mod.ExtraBody), ""); path != "" {
return fmt.Sprintf("extra_body carries the thinking-control key %q", path)
}
return ""
}

View file

@ -220,6 +220,12 @@ func LoadPipeline(path string, models *Models) (*Pipeline, error) {
} else if st.EscalateTo == st.Model {
bad("stage %q: escalate_to must differ from the primary model %q (a same-model hop is a guaranteed repeat)", st.Name, st.Model)
}
// D12 editor-pinned, enforced structurally: only the translator role may
// fall back to another model. An editor/other stage that escalated would
// drift its style/terms to a foreign model (2605.13368) — forbid it at load.
if st.Role != "translator" {
bad("stage %q: escalate_to is only allowed on a translator role (D12 editor-pinned — a %q stage must not fall back to a foreign model)", st.Name, st.Role)
}
}
// D4.1: channel-B (18+) isolation is enforced by TYPE — a permissive stage
// may only run on, and escalate to, a permissive provider.

View file

@ -33,6 +33,12 @@ import (
// bumps this constant.
const coverageGateVersion = "coverage-v1-naive-split-lower-bound"
// oracleDefaultCorridorLow is refusal_bench.py's fallback len_ratio LOWER bound for a
// source language without an explicit corridor (`EXPECT_LEN_RATIO.get(lang, (0.5,
// 3.0))`). The Go gate applies it too, so a book pair absent from len_ratio_bounds is
// checked exactly as the oracle would — not silently skipped (external-review parity).
const oracleDefaultCorridorLow = 0.5
// isSentenceTerminator is the branch-1 lookbehind class of refusal_bench's
// SENT_SPLIT_RE `[.!?…。!?]` (Latin + ellipsis + fullwidth CJK): a whitespace run
// after any of these splits a sentence.
@ -172,12 +178,17 @@ func coverageCheck(cfg config.CoverageGate, src, out, srcLang, dstLang string) c
if sentCov < cfg.SentCovMin {
flags = append(flags, fmt.Sprintf("sent_cov=%.2f<%.2f", sentCov, cfg.SentCovMin))
}
// len_ratio needs a corridor for this pair; without one we still apply sentence
// coverage (pair-independent) rather than silently pass everything.
// len_ratio corridor: the book's pair from config, else the ORACLE's fallback
// (0.5) so a pair without a configured corridor still gets the exact check the
// Python oracle applies (`EXPECT_LEN_RATIO.get(lang, (0.5, 3.0))`) — Go must not
// silently skip it (external-review parity finding). Only the lower bound is used
// (upper is out of scope, D12 Q3).
lo := oracleDefaultCorridorLow
if bounds, ok := cfg.LenRatio[langPairKey(srcLang, dstLang)]; ok && len(bounds) >= 1 {
if lo := bounds[0]; lenRatio < lo {
flags = append(flags, fmt.Sprintf("len_ratio=%.2f<%.2f", lenRatio, lo))
}
lo = bounds[0]
}
if lenRatio < lo {
flags = append(flags, fmt.Sprintf("len_ratio=%.2f<%.2f", lenRatio, lo))
}
if len(flags) > 0 {
res.cls = classification{FlagExcisionSuspect, strings.Join(flags, "; ")}

View file

@ -150,6 +150,23 @@ func TestCoverageCheckExcision(t *testing.T) {
}
}
// TestCoverageDefaultCorridorForUnknownPair pins external-review parity: a pair absent
// from len_ratio_bounds still gets the ORACLE's default corridor (0.5), not a silently
// skipped len check. sent_cov is kept at 1.0 so ONLY the default len_ratio can flag.
// Mutation: revert the default corridor and this goes green-to-red.
func TestCoverageDefaultCorridorForUnknownPair(t *testing.T) {
gate := config.CoverageGate{Enabled: true, SentCovMin: 0.75, MinChunkChars: 0} // no LenRatio map at all
src := strings.Repeat("가나다라마바사아자차카타파。", 3) // 3 ko sentences, long
out := "다. 라. 마." // 3 sentences (sent_cov 1.0) but tiny → len_ratio ≪ 0.5
res := coverageCheck(gate, src, out, "ko", "ru")
if res.cls.Reason != FlagExcisionSuspect || !strings.Contains(res.cls.Detail, "len_ratio") {
t.Fatalf("a no-corridor pair must apply the oracle default (0.5) len_ratio, got %+v", res)
}
if strings.Contains(res.cls.Detail, "sent_cov") {
t.Fatalf("sent_cov 1.0 must not flag here — the len_ratio default corridor is what fires: %q", res.cls.Detail)
}
}
func TestCoverageCheckApplicabilityAndCorridor(t *testing.T) {
// Below min_chunk_chars → gate is NOT applied (short-chunk ratios are noise).
short := coverageCheck(boevoyGate(500), "这是一个测试。", "x", "zh", "ru")

View file

@ -74,6 +74,17 @@ const (
FlagUpstreamNotOK FlagReason = "upstream_not_ok"
)
// classifierVersion versions the INTRINSIC classify() verdict logic — the refusal
// blacklist, the CJK-echo threshold (cjkEchoThreshold), the degeneration detector and
// the order in which they run. classify() resolves the ok↔flagged disposition and is
// re-run on resume over legacy / in-flight-crash checkpoints, so a change to any
// threshold (e.g. cjkEchoThreshold 0.15→0.20, a new refusal pattern) would otherwise
// keep the SAME snapshot id and silently re-verdict a resumed chunk — flagged→ok
// re-runs a skipped stage (fresh spend), ok→flagged burns a fresh escalation hop.
// Folded into the snapshot exactly like coverageGateVersion, so such a change is a
// loud --resnapshot, not a silent divergence (external-review finding).
const classifierVersion = "classify-v1-refusal+echo015+loop"
// decodeErrorFinish is the finish_reason the runner stores on a billed-but-
// unreadable 2xx (BilledDecodeError). classify() recognises it so a RESUMED
// decode checkpoint re-resolves to the same FlagDecodeError verdict the live

View file

@ -177,8 +177,9 @@ func (r *Runner) coverageSnapshot() coverageSnap {
// pure and deterministic over (source, output, finish), so a resumed checkpoint
// reproduces the identical verdict for free — the intrinsic part unconditionally, the
// gate part under the same coverage config (folded into the snapshot, so a gate change
// re-pins loudly). The gate compares the output against the ORIGINAL source (ch.Text)
// at EVERY stage, so excision introduced by the draft OR the editor is caught.
// re-pins loudly). The coverage gate runs ONLY on the TRANSLATOR role's output (vs the
// original source): a monolingual editor legitimately restructures sentences, so gating
// it against the source would false-flag a correct edit (see the role check below).
func (r *Runner) classifyOutput(role, source, output, finish string) classification {
cls := classify(classifyInput{Source: source, Output: output, Finish: finish, TargetLang: r.Book.TargetLang})
if !cls.ok() || !r.Pipeline.Gates.Coverage.Enabled {
@ -260,6 +261,12 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
// local-пути (находка внешнего ревью F2).
ProviderTemp float64 `json:"provider_temp,omitempty"`
ProviderMaxTok int `json:"provider_max_tok,omitempty"`
// ProviderModel — the local-kind backend tag the provider swaps onto the wire
// AFTER the request-hash (the actual model that answers). The MOST impactful
// local override, yet it was missing here while its weaker temp/max_tok siblings
// were folded: a local swap 8b→14b mid-book keeps the same snapID and resume
// serves the old model (external-review). Folded so it is a loud --resnapshot.
ProviderModel string `json:"provider_model,omitempty"`
// Capability — резолвнутая wire-форма модели (D3.1): budget-ключ,
// temperature-режим, reasoning-контроль. Меняет ТЕЛО запроса (max_tokens
// vs max_completion_tokens, отправлять ли temperature, thinking-выключа-
@ -284,6 +291,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
EscalateExtra json.RawMessage `json:"escalate_extra,omitempty"`
EscalateProviderTemp float64 `json:"escalate_provider_temp,omitempty"`
EscalateProviderMaxTok int `json:"escalate_provider_max_tok,omitempty"`
EscalateProviderModel string `json:"escalate_provider_model,omitempty"`
}
snap := struct {
BriefHash string `json:"brief_hash"`
@ -294,7 +302,11 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
// belongs in the snapshot as a loud invalidation (same class as
// estimator_version, applied to the regeneration axis).
MaxTokensPolicy string `json:"max_tokens_policy"`
PipelineCore string `json:"pipeline_core"`
// ClassifierVersion versions the intrinsic classify() verdict logic (thresholds
// + order), so a re-verdict on a resumed checkpoint is a loud --resnapshot, not
// a silent flagged↔ok divergence (external-review; symmetric to coverage).
ClassifierVersion string `json:"classifier_version"`
PipelineCore string `json:"pipeline_core"`
// Defaults влияют на maxTokens, а тот входит в request-hash: без них
// правка max_output_ratio молча инвалидировала бы все чекпоинты в
// обход snapshot-гейта (находка ревью).
@ -318,13 +330,14 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
Coverage coverageSnap `json:"coverage"`
Stages []stageSnap `json:"stages"`
}{
BriefHash: r.Book.BriefHash(),
ChunkerVersion: chunkerVersion,
EstimatorVersion: estimatorVersion,
MaxTokensPolicy: maxTokensPolicyVersion,
PipelineCore: r.Pipeline.Core,
MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio,
MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens,
BriefHash: r.Book.BriefHash(),
ChunkerVersion: chunkerVersion,
EstimatorVersion: estimatorVersion,
MaxTokensPolicy: maxTokensPolicyVersion,
ClassifierVersion: classifierVersion,
PipelineCore: r.Pipeline.Core,
MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio,
MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens,
ContextAssembly: contextSnap{
GlossaryInjection: r.Pipeline.Context.GlossaryInjection,
GlossaryTokenBudget: r.Pipeline.Context.GlossaryTokenBudget,
@ -342,7 +355,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
Temperature: st.Temperature, Reasoning: st.Reasoning,
}
if prov, ok := r.Models.Providers[r.Models.Models[st.Model].Provider]; ok {
ss.ProviderTemp, ss.ProviderMaxTok = prov.Temperature, prov.MaxTokens
ss.ProviderTemp, ss.ProviderMaxTok, ss.ProviderModel = prov.Temperature, prov.MaxTokens, prov.Model
}
if extra := r.Models.Models[st.Model].ExtraBody; len(extra) > 0 {
raw, merr := json.Marshal(extra)
@ -370,7 +383,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) {
}
ss.EscalateCapability = escCap
if prov, ok := r.Models.Providers[r.Models.Models[st.EscalateTo].Provider]; ok {
ss.EscalateProviderTemp, ss.EscalateProviderMaxTok = prov.Temperature, prov.MaxTokens
ss.EscalateProviderTemp, ss.EscalateProviderMaxTok, ss.EscalateProviderModel = prov.Temperature, prov.MaxTokens, prov.Model
}
if extra := r.Models.Models[st.EscalateTo].ExtraBody; len(extra) > 0 {
raw, merr := json.Marshal(extra)

View file

@ -1350,6 +1350,102 @@ func TestRunnerEscalationEntersSnapshot(t *testing.T) {
}
}
// D12 editor-pinned, enforced structurally (external-review [4]): escalate_to on a
// NON-translator role is rejected at load — an editor must not fall back to a foreign
// model. Mutation: drop the role check and this loads instead of failing.
func TestRunnerRejectsEscalateToOnNonTranslator(t *testing.T) {
rec := &reqRec{}
srv := newJSONProvider(rec, echoOrClean)
defer srv.Close()
bookPath := setupEscalationProject(t, srv.URL, 1.0, false, "")
pipePath := filepath.Join(filepath.Dir(bookPath), "pipeline.yaml")
raw, err := os.ReadFile(pipePath)
if err != nil {
t.Fatal(err)
}
// Put escalate_to on the EDITOR stage (role=editor) — must be rejected.
patched := strings.Replace(string(raw),
`{ name: edit, role: editor, model: fake-model, prompt: prompts/editor.md, prompt_version: v-test, temperature: 0.4, reasoning: "off" }`,
`{ name: edit, role: editor, model: fake-model, prompt: prompts/editor.md, prompt_version: v-test, temperature: 0.4, reasoning: "off", escalate_to: fake-fallback }`, 1)
if patched == string(raw) {
t.Fatal("failed to inject escalate_to onto the editor stage")
}
writeFile(t, pipePath, patched)
if _, err := NewRunner(bookPath, obs.NewLogger()); err == nil || !strings.Contains(err.Error(), "translator") {
t.Fatalf("escalate_to on a non-translator role must fail-fast (D12 editor-pinned), got: %v", err)
}
}
// The intrinsic classify() version AND the local backend tag are folded into the
// snapshot (external-review [1]/[2]): a threshold or backend-tag change is a loud
// --resnapshot, not a silent re-verdict / stale-model resume. Mutation: drop either
// fold and the payload no longer carries it.
func TestRunnerSnapshotFoldsClassifierAndLocalTag(t *testing.T) {
dir := t.TempDir()
writeFile(t, filepath.Join(dir, "prompts", "translator.md"),
"Переводи с {{source_lang}} на {{target_lang}}.\n---USER---\n{{text}}")
writeFile(t, filepath.Join(dir, "models.yaml"), fmt.Sprintf(`
prices_checked: %q
default_model: anchor
providers:
loc:
kind: local
base_url: http://127.0.0.1:11434/v1
model: my-backend-tag-8b
max_tokens: 8192
cloud:
kind: openai
base_url: http://x
models:
anchor:
provider: cloud
price: { input_per_m: 1, cached_per_m: 0, cache_write_per_m: 0, output_per_m: 2 }
local-m:
provider: loc
price: { input_per_m: 0, cached_per_m: 0, cache_write_per_m: 0, output_per_m: 0 }
`, time.Now().UTC().Format("2006-01-02")))
writeFile(t, filepath.Join(dir, "pipeline.yaml"), `
core: C1
version: 1
defaults: { max_output_ratio: 2.0, min_max_tokens: 512 }
retries: { regenerate_before_escalate: 0 }
stages:
- { name: draft, role: translator, model: local-m, prompt: prompts/translator.md, prompt_version: v-test, temperature: 0.3, reasoning: "off" }
`)
writeFile(t, filepath.Join(dir, "source.txt"), "猫。")
writeFile(t, filepath.Join(dir, "book.yaml"), `
book_id: b
title: T
source_lang: zh
target_lang: ru
genre: g
audience: a
venuti: 0.5
honorifics: keep
transcription: pinyin
footnotes: minimal
pipeline: pipeline.yaml
models: models.yaml
source_file: source.txt
ceilings: { book_usd: 1.0, day_usd: 1.0 }
`)
r := newRunner(t, filepath.Join(dir, "book.yaml"))
defer r.Close()
_, payload, err := r.snapshotID()
if err != nil {
t.Fatal(err)
}
if !strings.Contains(payload, classifierVersion) {
t.Fatalf("snapshot must fold the classifier version %q, got: %s", classifierVersion, payload)
}
if !strings.Contains(payload, "my-backend-tag-8b") {
t.Fatalf("snapshot must fold the local backend tag (provider_model), got: %s", payload)
}
}
// D4.1: channel-B (18+) isolation is enforced by TYPE — a channel=adult stage on a
// NON-permissive provider is rejected at load, never a silent fall-through.
func TestRunnerChannelBRequiresPermissive(t *testing.T) {

View file

@ -368,6 +368,18 @@ keep-alive (Ф12), инъекция глоссария (`selective`), пор
> - **`escalation.budget_usd`:** механика (opt-in soft-cap, 0=выкл) верна; число — $-потолок под маржу владельца + пересчёт полигона (D11), не «≤2×»; 0=выкл — норм дев-дефолт.
> - **D12-хвосты, приоритет:** **память v2** (реестр `06` готов — главный блокер качества) → **`tmctl status`+manifest** (владелец просил прогресс, скоуп мал, `chunk_status` есть) → **F3** (консервативный settle — честный интерим). Порядок предложишь сам.
### 2026-07-05 — Веха 2.5: закрытие находок внешнего ревью (7/7)
Все 7 подтверждённых находок закрыты, каждый фикс мутационно-проверен (реверт → падает именно его тест). Классический слепок автора (снапшот-дисциплину применил к `coverageGateVersion`, но не к 3 смежным determinator'ам) — закрыт симметрично.
**3 major:** [1] `classifierVersion` свёрнут в snapshot рядом с `coverageGateVersion` (правка порога `classify` = громкий `--resnapshot`, не тихий re-verdict). [2] local backend-тег `prov.Model` (+ `EscalateProviderModel`) свёрнут в `stageSnap` (своп 8b→14b посреди книги теперь инвалидирует чекпоинты). [3] echo-мина: плоский скан 3 ключей → **рекурсивный** `findThinkingControlKey` — ловит вложенные формы (`chat_template_kwargs.thinking` = документированное отключение DeepSeek-V3.1+, самая вероятная реальная).
**4 minor:** [4] `escalate_to` разрешён только на `role=translator` (структурное принуждение D12 editor-pinned). [5] coverage: пара без коридора теперь получает дефолт оракула (0.5) вместо пропуска len-проверки (восстановлен claim «бит-в-бит»). [6] телеметрия эскалации на flagged-resume — **отложена ревьюером** до `tmctl status` (нужна колонка escalation в chunk_status; деньги/флаг верны, теряется только in-memory Escalated). [7] врущий докстринг `classifyOutput` поправлен (гейт только translator, не «каждая стадия»).
Регресс-тесты: `chat_template_kwargs.thinking`-кейс, no-corridor default-коридор, escalate_to-на-editor→fail, snapshot-fold `classifierVersion`+local-тег. Полный набор зелёный `-race`, `tmctl report` $0. **deepseek-v4-pro цена подтверждена оркестратором** ($0.435/$0.87/$0.003625) — разблокирует wiring канала-A эскалации.
**Веха 2.5 ПОЛНОСТЬЮ закрыта** (реализация + селфревью 14 + внешнее ревью 7). Дальше — D12-хвосты по приоритету оркестратора (память v2 → tmctl status → F3).
## Полигон
(секция параллельной сессии — записи добавлять сюда)
@ -504,3 +516,5 @@ keep-alive (Ф12), инъекция глоссария (`selective`), пор
> **[ПИЛОТУ]** M1 (in-context демонстрации, серая зона) фолдится в тот же WMT25-3-режимный eval, что и банк, **+ стилевая ось** (win-rate человека/сильного судьи): критерий — бьёт ли базис по стилю БЕЗ потери консистентности/fidelity и без роста дистрактор-вреда. Индикатив этой сессии: по консистентности не бьёт; стиль не измерен.
> **[ГИГИЕНА]** Модель полигона `qwen3-abliterated:30b-a3b` (6.4 ГБ VRAM) **не выселялась** — GPU-проба шла в свободном CUDA-окне (WSL2 спилит в shared RAM), после — cleanup. Артефакты: `eval/adaptive_probe.py`, `eval/adaptive_incontext.py` (+ `eval/data/*.json`), новые доки — только `research/14`. Изолированный CUDA-env (Pascal-проба) — в scratchpad, не в репозитории. Полный потолок QLoRA-7B+reranker на **чистых** 8ГБ — за скоординированным стенд-окном (не мерил, чтоб не выселять 30b); на вердикт не влияет (кандидата нет).
> **[ДОБАВЛЕНО, 05.07 — «долбить детерминированную сторону»]** По запросу владельца (раз гибрид отклонён — максимизировать коэффициент канон-консистентности детерминированно) — замер рычагов L1→L4 на ТРУДНОМ чанке (`eval/adaptive_levers.py`, 11 сущностей, deepseek+grok), research/14 §9. **Порядок бэкенду:** **L1 post-check ЛЕММАТИЗИРУЮЩИЙ (pymorphy3), не регэксп** — ПЕРВЫМ (наивный регэксп даёт **1836% ЛОЖНЫХ флагов** на трудном чанке; pymorphy сворачивает и OOV-транслит Вана→ван; количественная валидация несущего риска research/13 §2). **L2 recall — нормализация ПОЛНАЯ+тестируемая (OpenCC, не мини-карта: живой A4-баг 爺→爷 молча промахнул 赵太爷) + alias-граф + sticky** (recall 0.00→0.50→0.75). **L3 re-ask** (deepseek 0.82→0.91). **L4 таблица Палладия (B6) — потолок ПРАВИЛЬНОСТИ.** Побочно: 钱 (односимвольный гомограф Цянь/деньги) упорно промахивается → **min_key_len/запрет одиночных ключей A3 обязателен.** Поправка §2: soft-глоссарий на лёгком чанке = **~1.0** (0.875 был регэксп-артефакт), demos не бьют — потолок. Артефакты: `eval/adaptive_reask.py`, `eval/adaptive_levers.py`. Инстинкт «тюнить ранкер» — легитимен только во 2-м эшелоне (bge-m3 без разделяющего порога), не в костяке.