diff --git a/backend/internal/config/echo_mine_test.go b/backend/internal/config/echo_mine_test.go index e4641f0..e87dad7 100644 --- a/backend/internal/config/echo_mine_test.go +++ b/backend/internal/config/echo_mine_test.go @@ -81,6 +81,15 @@ models: modelExtra: "\n extra_body: { thinking: { type: disabled } }", wantErr: true, }, + { + // The NESTED form: chat_template_kwargs.thinking:false is DeepSeek's + // documented V3.1+ disable — the most likely real arming form. A flat + // top-level scan misses it; the recursive scan must catch it (ext-review). + name: "armed via nested chat_template_kwargs.thinking (DeepSeek V3.1+)", + echoProne: true, + modelExtra: "\n extra_body: { chat_template_kwargs: { thinking: false } }", + wantErr: true, + }, { // reasoning_effort at off also pushes thinking off the default. name: "armed via capabilities.reasoning.effort", diff --git a/backend/internal/config/models.go b/backend/internal/config/models.go index 554d0ca..64a121b 100644 --- a/backend/internal/config/models.go +++ b/backend/internal/config/models.go @@ -270,22 +270,50 @@ func (m *Models) ResolveCapability(modelName string) llm.Capability { return resolved } -// thinkingControlExtraKeys are model-level extra_body keys whose sole purpose is -// to toggle a provider's thinking on the wire (GLM/DeepSeek {"thinking":…}, Qwen -// {"enable_thinking":…}, a raw {"reasoning_effort":…}). On an echo-prone provider -// their mere PRESENCE is the mine, whatever the value: thinking must stay at the -// provider default, and these keys exist only to move it off that default. -var thinkingControlExtraKeys = []string{"thinking", "enable_thinking", "reasoning_effort"} +// thinkingControlExtraKeys are keys whose purpose is to toggle a provider's thinking +// on the wire — top-level (GLM/DeepSeek {"thinking":…}, Qwen {"enable_thinking":…}, a +// raw {"reasoning_effort":…}) OR nested (DeepSeek-V3.1+ disables via +// chat_template_kwargs.thinking:false). On an echo-prone provider their mere PRESENCE, +// at any depth, is the mine, whatever the value: thinking must stay at the provider +// default, and these keys exist only to move it off that default. +var thinkingControlExtraKeys = map[string]bool{ + "thinking": true, "enable_thinking": true, "reasoning_effort": true, "reasoning": true, +} + +// findThinkingControlKey RECURSIVELY scans an extra_body value for any thinking-control +// key, returning its dotted path (e.g. "chat_template_kwargs.thinking") or "". The +// recursion is what catches nested disable forms a flat top-level check would miss +// (external-review finding — chat_template_kwargs.thinking is DeepSeek's documented +// V3.1+ disable, the most likely real arming form). +func findThinkingControlKey(v any, path string) string { + m, ok := v.(map[string]any) + if !ok { + return "" + } + for k, sub := range m { + here := k + if path != "" { + here = path + "." + k + } + if thinkingControlExtraKeys[k] { + return here + } + if found := findThinkingControlKey(sub, here); found != "" { + return found + } + } + return "" +} // echoMineViolation reports, for one model, whether its RESOLVED wire shape would // suppress thinking at reasoning=off on an echo-prone provider — the DeepSeek echo -// mine (§2). It returns a human cause string, or "" when safe. Both injection -// surfaces are covered: the capabilities.reasoning switch (extra_body_disable -// merges a thinking-disable at off; effort sends reasoning_effort at off — both -// push thinking off the default) AND a thinking-control key smuggled through the -// model's top-level extra_body (merged into the wire body by openAIRequest). A -// non-echo-prone provider (GLM: its editor input is a Russian draft, nothing CJK -// to echo) is never gated, so its {thinking:{type:disabled}} stays legal. +// mine (§2). It returns a human cause string, or "" when safe. It covers both injection +// surfaces: the capabilities.reasoning switch (extra_body_disable merges a +// thinking-disable at off; effort sends reasoning_effort at off — both push thinking +// off the default) AND a thinking-control key anywhere in the model's extra_body, +// TOP-LEVEL OR NESTED (merged into the wire body by openAIRequest). A non-echo-prone +// provider (GLM: its editor input is a Russian draft, nothing CJK to echo) is never +// gated, so its {thinking:{type:disabled}} stays legal. func (m *Models) echoMineViolation(name string) string { mod := m.Models[name] prov, ok := m.Providers[mod.Provider] @@ -298,10 +326,8 @@ func (m *Models) echoMineViolation(name string) string { case llm.ReasoningEffortField: return "capabilities.reasoning.control=effort would send reasoning_effort at reasoning=off, suppressing thinking" } - for _, k := range thinkingControlExtraKeys { - if _, present := mod.ExtraBody[k]; present { - return fmt.Sprintf("extra_body carries the thinking-control key %q", k) - } + if path := findThinkingControlKey(map[string]any(mod.ExtraBody), ""); path != "" { + return fmt.Sprintf("extra_body carries the thinking-control key %q", path) } return "" } diff --git a/backend/internal/config/pipeline.go b/backend/internal/config/pipeline.go index 2ce103a..667a8a7 100644 --- a/backend/internal/config/pipeline.go +++ b/backend/internal/config/pipeline.go @@ -220,6 +220,12 @@ func LoadPipeline(path string, models *Models) (*Pipeline, error) { } else if st.EscalateTo == st.Model { bad("stage %q: escalate_to must differ from the primary model %q (a same-model hop is a guaranteed repeat)", st.Name, st.Model) } + // D12 editor-pinned, enforced structurally: only the translator role may + // fall back to another model. An editor/other stage that escalated would + // drift its style/terms to a foreign model (2605.13368) — forbid it at load. + if st.Role != "translator" { + bad("stage %q: escalate_to is only allowed on a translator role (D12 editor-pinned — a %q stage must not fall back to a foreign model)", st.Name, st.Role) + } } // D4.1: channel-B (18+) isolation is enforced by TYPE — a permissive stage // may only run on, and escalate to, a permissive provider. diff --git a/backend/internal/pipeline/coverage.go b/backend/internal/pipeline/coverage.go index 5cf0d9d..292e276 100644 --- a/backend/internal/pipeline/coverage.go +++ b/backend/internal/pipeline/coverage.go @@ -33,6 +33,12 @@ import ( // bumps this constant. const coverageGateVersion = "coverage-v1-naive-split-lower-bound" +// oracleDefaultCorridorLow is refusal_bench.py's fallback len_ratio LOWER bound for a +// source language without an explicit corridor (`EXPECT_LEN_RATIO.get(lang, (0.5, +// 3.0))`). The Go gate applies it too, so a book pair absent from len_ratio_bounds is +// checked exactly as the oracle would — not silently skipped (external-review parity). +const oracleDefaultCorridorLow = 0.5 + // isSentenceTerminator is the branch-1 lookbehind class of refusal_bench's // SENT_SPLIT_RE `[.!?…。!?]` (Latin + ellipsis + fullwidth CJK): a whitespace run // after any of these splits a sentence. @@ -172,12 +178,17 @@ func coverageCheck(cfg config.CoverageGate, src, out, srcLang, dstLang string) c if sentCov < cfg.SentCovMin { flags = append(flags, fmt.Sprintf("sent_cov=%.2f<%.2f", sentCov, cfg.SentCovMin)) } - // len_ratio needs a corridor for this pair; without one we still apply sentence - // coverage (pair-independent) rather than silently pass everything. + // len_ratio corridor: the book's pair from config, else the ORACLE's fallback + // (0.5) so a pair without a configured corridor still gets the exact check the + // Python oracle applies (`EXPECT_LEN_RATIO.get(lang, (0.5, 3.0))`) — Go must not + // silently skip it (external-review parity finding). Only the lower bound is used + // (upper is out of scope, D12 Q3). + lo := oracleDefaultCorridorLow if bounds, ok := cfg.LenRatio[langPairKey(srcLang, dstLang)]; ok && len(bounds) >= 1 { - if lo := bounds[0]; lenRatio < lo { - flags = append(flags, fmt.Sprintf("len_ratio=%.2f<%.2f", lenRatio, lo)) - } + lo = bounds[0] + } + if lenRatio < lo { + flags = append(flags, fmt.Sprintf("len_ratio=%.2f<%.2f", lenRatio, lo)) } if len(flags) > 0 { res.cls = classification{FlagExcisionSuspect, strings.Join(flags, "; ")} diff --git a/backend/internal/pipeline/coverage_test.go b/backend/internal/pipeline/coverage_test.go index 9fbb99c..fddda87 100644 --- a/backend/internal/pipeline/coverage_test.go +++ b/backend/internal/pipeline/coverage_test.go @@ -150,6 +150,23 @@ func TestCoverageCheckExcision(t *testing.T) { } } +// TestCoverageDefaultCorridorForUnknownPair pins external-review parity: a pair absent +// from len_ratio_bounds still gets the ORACLE's default corridor (0.5), not a silently +// skipped len check. sent_cov is kept at 1.0 so ONLY the default len_ratio can flag. +// Mutation: revert the default corridor and this goes green-to-red. +func TestCoverageDefaultCorridorForUnknownPair(t *testing.T) { + gate := config.CoverageGate{Enabled: true, SentCovMin: 0.75, MinChunkChars: 0} // no LenRatio map at all + src := strings.Repeat("가나다라마바사아자차카타파。", 3) // 3 ko sentences, long + out := "다. 라. 마." // 3 sentences (sent_cov 1.0) but tiny → len_ratio ≪ 0.5 + res := coverageCheck(gate, src, out, "ko", "ru") + if res.cls.Reason != FlagExcisionSuspect || !strings.Contains(res.cls.Detail, "len_ratio") { + t.Fatalf("a no-corridor pair must apply the oracle default (0.5) len_ratio, got %+v", res) + } + if strings.Contains(res.cls.Detail, "sent_cov") { + t.Fatalf("sent_cov 1.0 must not flag here — the len_ratio default corridor is what fires: %q", res.cls.Detail) + } +} + func TestCoverageCheckApplicabilityAndCorridor(t *testing.T) { // Below min_chunk_chars → gate is NOT applied (short-chunk ratios are noise). short := coverageCheck(boevoyGate(500), "这是一个测试。", "x", "zh", "ru") diff --git a/backend/internal/pipeline/disposition.go b/backend/internal/pipeline/disposition.go index 24028d5..cae7dc3 100644 --- a/backend/internal/pipeline/disposition.go +++ b/backend/internal/pipeline/disposition.go @@ -74,6 +74,17 @@ const ( FlagUpstreamNotOK FlagReason = "upstream_not_ok" ) +// classifierVersion versions the INTRINSIC classify() verdict logic — the refusal +// blacklist, the CJK-echo threshold (cjkEchoThreshold), the degeneration detector and +// the order in which they run. classify() resolves the ok↔flagged disposition and is +// re-run on resume over legacy / in-flight-crash checkpoints, so a change to any +// threshold (e.g. cjkEchoThreshold 0.15→0.20, a new refusal pattern) would otherwise +// keep the SAME snapshot id and silently re-verdict a resumed chunk — flagged→ok +// re-runs a skipped stage (fresh spend), ok→flagged burns a fresh escalation hop. +// Folded into the snapshot exactly like coverageGateVersion, so such a change is a +// loud --resnapshot, not a silent divergence (external-review finding). +const classifierVersion = "classify-v1-refusal+echo015+loop" + // decodeErrorFinish is the finish_reason the runner stores on a billed-but- // unreadable 2xx (BilledDecodeError). classify() recognises it so a RESUMED // decode checkpoint re-resolves to the same FlagDecodeError verdict the live diff --git a/backend/internal/pipeline/runner.go b/backend/internal/pipeline/runner.go index a4f6cb6..d240748 100644 --- a/backend/internal/pipeline/runner.go +++ b/backend/internal/pipeline/runner.go @@ -177,8 +177,9 @@ func (r *Runner) coverageSnapshot() coverageSnap { // pure and deterministic over (source, output, finish), so a resumed checkpoint // reproduces the identical verdict for free — the intrinsic part unconditionally, the // gate part under the same coverage config (folded into the snapshot, so a gate change -// re-pins loudly). The gate compares the output against the ORIGINAL source (ch.Text) -// at EVERY stage, so excision introduced by the draft OR the editor is caught. +// re-pins loudly). The coverage gate runs ONLY on the TRANSLATOR role's output (vs the +// original source): a monolingual editor legitimately restructures sentences, so gating +// it against the source would false-flag a correct edit (see the role check below). func (r *Runner) classifyOutput(role, source, output, finish string) classification { cls := classify(classifyInput{Source: source, Output: output, Finish: finish, TargetLang: r.Book.TargetLang}) if !cls.ok() || !r.Pipeline.Gates.Coverage.Enabled { @@ -260,6 +261,12 @@ func (r *Runner) snapshotID() (id, payload string, err error) { // local-пути (находка внешнего ревью F2). ProviderTemp float64 `json:"provider_temp,omitempty"` ProviderMaxTok int `json:"provider_max_tok,omitempty"` + // ProviderModel — the local-kind backend tag the provider swaps onto the wire + // AFTER the request-hash (the actual model that answers). The MOST impactful + // local override, yet it was missing here while its weaker temp/max_tok siblings + // were folded: a local swap 8b→14b mid-book keeps the same snapID and resume + // serves the old model (external-review). Folded so it is a loud --resnapshot. + ProviderModel string `json:"provider_model,omitempty"` // Capability — резолвнутая wire-форма модели (D3.1): budget-ключ, // temperature-режим, reasoning-контроль. Меняет ТЕЛО запроса (max_tokens // vs max_completion_tokens, отправлять ли temperature, thinking-выключа- @@ -284,6 +291,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) { EscalateExtra json.RawMessage `json:"escalate_extra,omitempty"` EscalateProviderTemp float64 `json:"escalate_provider_temp,omitempty"` EscalateProviderMaxTok int `json:"escalate_provider_max_tok,omitempty"` + EscalateProviderModel string `json:"escalate_provider_model,omitempty"` } snap := struct { BriefHash string `json:"brief_hash"` @@ -294,7 +302,11 @@ func (r *Runner) snapshotID() (id, payload string, err error) { // belongs in the snapshot as a loud invalidation (same class as // estimator_version, applied to the regeneration axis). MaxTokensPolicy string `json:"max_tokens_policy"` - PipelineCore string `json:"pipeline_core"` + // ClassifierVersion versions the intrinsic classify() verdict logic (thresholds + // + order), so a re-verdict on a resumed checkpoint is a loud --resnapshot, not + // a silent flagged↔ok divergence (external-review; symmetric to coverage). + ClassifierVersion string `json:"classifier_version"` + PipelineCore string `json:"pipeline_core"` // Defaults влияют на maxTokens, а тот входит в request-hash: без них // правка max_output_ratio молча инвалидировала бы все чекпоинты в // обход snapshot-гейта (находка ревью). @@ -318,13 +330,14 @@ func (r *Runner) snapshotID() (id, payload string, err error) { Coverage coverageSnap `json:"coverage"` Stages []stageSnap `json:"stages"` }{ - BriefHash: r.Book.BriefHash(), - ChunkerVersion: chunkerVersion, - EstimatorVersion: estimatorVersion, - MaxTokensPolicy: maxTokensPolicyVersion, - PipelineCore: r.Pipeline.Core, - MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio, - MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens, + BriefHash: r.Book.BriefHash(), + ChunkerVersion: chunkerVersion, + EstimatorVersion: estimatorVersion, + MaxTokensPolicy: maxTokensPolicyVersion, + ClassifierVersion: classifierVersion, + PipelineCore: r.Pipeline.Core, + MaxOutputRatio: r.Pipeline.Defaults.MaxOutputRatio, + MinMaxTokens: r.Pipeline.Defaults.MinMaxTokens, ContextAssembly: contextSnap{ GlossaryInjection: r.Pipeline.Context.GlossaryInjection, GlossaryTokenBudget: r.Pipeline.Context.GlossaryTokenBudget, @@ -342,7 +355,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) { Temperature: st.Temperature, Reasoning: st.Reasoning, } if prov, ok := r.Models.Providers[r.Models.Models[st.Model].Provider]; ok { - ss.ProviderTemp, ss.ProviderMaxTok = prov.Temperature, prov.MaxTokens + ss.ProviderTemp, ss.ProviderMaxTok, ss.ProviderModel = prov.Temperature, prov.MaxTokens, prov.Model } if extra := r.Models.Models[st.Model].ExtraBody; len(extra) > 0 { raw, merr := json.Marshal(extra) @@ -370,7 +383,7 @@ func (r *Runner) snapshotID() (id, payload string, err error) { } ss.EscalateCapability = escCap if prov, ok := r.Models.Providers[r.Models.Models[st.EscalateTo].Provider]; ok { - ss.EscalateProviderTemp, ss.EscalateProviderMaxTok = prov.Temperature, prov.MaxTokens + ss.EscalateProviderTemp, ss.EscalateProviderMaxTok, ss.EscalateProviderModel = prov.Temperature, prov.MaxTokens, prov.Model } if extra := r.Models.Models[st.EscalateTo].ExtraBody; len(extra) > 0 { raw, merr := json.Marshal(extra) diff --git a/backend/internal/pipeline/runner_test.go b/backend/internal/pipeline/runner_test.go index ea9e725..c00c45b 100644 --- a/backend/internal/pipeline/runner_test.go +++ b/backend/internal/pipeline/runner_test.go @@ -1350,6 +1350,102 @@ func TestRunnerEscalationEntersSnapshot(t *testing.T) { } } +// D12 editor-pinned, enforced structurally (external-review [4]): escalate_to on a +// NON-translator role is rejected at load — an editor must not fall back to a foreign +// model. Mutation: drop the role check and this loads instead of failing. +func TestRunnerRejectsEscalateToOnNonTranslator(t *testing.T) { + rec := &reqRec{} + srv := newJSONProvider(rec, echoOrClean) + defer srv.Close() + bookPath := setupEscalationProject(t, srv.URL, 1.0, false, "") + + pipePath := filepath.Join(filepath.Dir(bookPath), "pipeline.yaml") + raw, err := os.ReadFile(pipePath) + if err != nil { + t.Fatal(err) + } + // Put escalate_to on the EDITOR stage (role=editor) — must be rejected. + patched := strings.Replace(string(raw), + `{ name: edit, role: editor, model: fake-model, prompt: prompts/editor.md, prompt_version: v-test, temperature: 0.4, reasoning: "off" }`, + `{ name: edit, role: editor, model: fake-model, prompt: prompts/editor.md, prompt_version: v-test, temperature: 0.4, reasoning: "off", escalate_to: fake-fallback }`, 1) + if patched == string(raw) { + t.Fatal("failed to inject escalate_to onto the editor stage") + } + writeFile(t, pipePath, patched) + + if _, err := NewRunner(bookPath, obs.NewLogger()); err == nil || !strings.Contains(err.Error(), "translator") { + t.Fatalf("escalate_to on a non-translator role must fail-fast (D12 editor-pinned), got: %v", err) + } +} + +// The intrinsic classify() version AND the local backend tag are folded into the +// snapshot (external-review [1]/[2]): a threshold or backend-tag change is a loud +// --resnapshot, not a silent re-verdict / stale-model resume. Mutation: drop either +// fold and the payload no longer carries it. +func TestRunnerSnapshotFoldsClassifierAndLocalTag(t *testing.T) { + dir := t.TempDir() + writeFile(t, filepath.Join(dir, "prompts", "translator.md"), + "Переводи с {{source_lang}} на {{target_lang}}.\n---USER---\n{{text}}") + writeFile(t, filepath.Join(dir, "models.yaml"), fmt.Sprintf(` +prices_checked: %q +default_model: anchor +providers: + loc: + kind: local + base_url: http://127.0.0.1:11434/v1 + model: my-backend-tag-8b + max_tokens: 8192 + cloud: + kind: openai + base_url: http://x +models: + anchor: + provider: cloud + price: { input_per_m: 1, cached_per_m: 0, cache_write_per_m: 0, output_per_m: 2 } + local-m: + provider: loc + price: { input_per_m: 0, cached_per_m: 0, cache_write_per_m: 0, output_per_m: 0 } +`, time.Now().UTC().Format("2006-01-02"))) + writeFile(t, filepath.Join(dir, "pipeline.yaml"), ` +core: C1 +version: 1 +defaults: { max_output_ratio: 2.0, min_max_tokens: 512 } +retries: { regenerate_before_escalate: 0 } +stages: + - { name: draft, role: translator, model: local-m, prompt: prompts/translator.md, prompt_version: v-test, temperature: 0.3, reasoning: "off" } +`) + writeFile(t, filepath.Join(dir, "source.txt"), "猫。") + writeFile(t, filepath.Join(dir, "book.yaml"), ` +book_id: b +title: T +source_lang: zh +target_lang: ru +genre: g +audience: a +venuti: 0.5 +honorifics: keep +transcription: pinyin +footnotes: minimal +pipeline: pipeline.yaml +models: models.yaml +source_file: source.txt +ceilings: { book_usd: 1.0, day_usd: 1.0 } +`) + + r := newRunner(t, filepath.Join(dir, "book.yaml")) + defer r.Close() + _, payload, err := r.snapshotID() + if err != nil { + t.Fatal(err) + } + if !strings.Contains(payload, classifierVersion) { + t.Fatalf("snapshot must fold the classifier version %q, got: %s", classifierVersion, payload) + } + if !strings.Contains(payload, "my-backend-tag-8b") { + t.Fatalf("snapshot must fold the local backend tag (provider_model), got: %s", payload) + } +} + // D4.1: channel-B (18+) isolation is enforced by TYPE — a channel=adult stage on a // NON-permissive provider is rejected at load, never a silent fall-through. func TestRunnerChannelBRequiresPermissive(t *testing.T) { diff --git a/docs/PROGRESS.md b/docs/PROGRESS.md index eb4f6eb..d47b7f5 100644 --- a/docs/PROGRESS.md +++ b/docs/PROGRESS.md @@ -368,6 +368,18 @@ keep-alive (Ф1–2), инъекция глоссария (`selective`), пор > - **`escalation.budget_usd`:** механика (opt-in soft-cap, 0=выкл) верна; число — $-потолок под маржу владельца + пересчёт полигона (D11), не «≤2×»; 0=выкл — норм дев-дефолт. > - **D12-хвосты, приоритет:** **память v2** (реестр `06` готов — главный блокер качества) → **`tmctl status`+manifest** (владелец просил прогресс, скоуп мал, `chunk_status` есть) → **F3** (консервативный settle — честный интерим). Порядок предложишь сам. +### 2026-07-05 — Веха 2.5: закрытие находок внешнего ревью (7/7) + +Все 7 подтверждённых находок закрыты, каждый фикс мутационно-проверен (реверт → падает именно его тест). Классический слепок автора (снапшот-дисциплину применил к `coverageGateVersion`, но не к 3 смежным determinator'ам) — закрыт симметрично. + +**3 major:** [1] `classifierVersion` свёрнут в snapshot рядом с `coverageGateVersion` (правка порога `classify` = громкий `--resnapshot`, не тихий re-verdict). [2] local backend-тег `prov.Model` (+ `EscalateProviderModel`) свёрнут в `stageSnap` (своп 8b→14b посреди книги теперь инвалидирует чекпоинты). [3] echo-мина: плоский скан 3 ключей → **рекурсивный** `findThinkingControlKey` — ловит вложенные формы (`chat_template_kwargs.thinking` = документированное отключение DeepSeek-V3.1+, самая вероятная реальная). + +**4 minor:** [4] `escalate_to` разрешён только на `role=translator` (структурное принуждение D12 editor-pinned). [5] coverage: пара без коридора теперь получает дефолт оракула (0.5) вместо пропуска len-проверки (восстановлен claim «бит-в-бит»). [6] телеметрия эскалации на flagged-resume — **отложена ревьюером** до `tmctl status` (нужна колонка escalation в chunk_status; деньги/флаг верны, теряется только in-memory Escalated). [7] врущий докстринг `classifyOutput` поправлен (гейт только translator, не «каждая стадия»). + +Регресс-тесты: `chat_template_kwargs.thinking`-кейс, no-corridor default-коридор, escalate_to-на-editor→fail, snapshot-fold `classifierVersion`+local-тег. Полный набор зелёный `-race`, `tmctl report` $0. **deepseek-v4-pro цена подтверждена оркестратором** ($0.435/$0.87/$0.003625) — разблокирует wiring канала-A эскалации. + +**Веха 2.5 ПОЛНОСТЬЮ закрыта** (реализация + селфревью 14 + внешнее ревью 7). Дальше — D12-хвосты по приоритету оркестратора (память v2 → tmctl status → F3). + ## Полигон (секция параллельной сессии — записи добавлять сюда) @@ -504,3 +516,5 @@ keep-alive (Ф1–2), инъекция глоссария (`selective`), пор > **[ПИЛОТУ]** M1 (in-context демонстрации, серая зона) фолдится в тот же WMT25-3-режимный eval, что и банк, **+ стилевая ось** (win-rate человека/сильного судьи): критерий — бьёт ли базис по стилю БЕЗ потери консистентности/fidelity и без роста дистрактор-вреда. Индикатив этой сессии: по консистентности не бьёт; стиль не измерен. > **[ГИГИЕНА]** Модель полигона `qwen3-abliterated:30b-a3b` (6.4 ГБ VRAM) **не выселялась** — GPU-проба шла в свободном CUDA-окне (WSL2 спилит в shared RAM), после — cleanup. Артефакты: `eval/adaptive_probe.py`, `eval/adaptive_incontext.py` (+ `eval/data/*.json`), новые доки — только `research/14`. Изолированный CUDA-env (Pascal-проба) — в scratchpad, не в репозитории. Полный потолок QLoRA-7B+reranker на **чистых** 8ГБ — за скоординированным стенд-окном (не мерил, чтоб не выселять 30b); на вердикт не влияет (кандидата нет). + +> **[ДОБАВЛЕНО, 05.07 — «долбить детерминированную сторону»]** По запросу владельца (раз гибрид отклонён — максимизировать коэффициент канон-консистентности детерминированно) — замер рычагов L1→L4 на ТРУДНОМ чанке (`eval/adaptive_levers.py`, 11 сущностей, deepseek+grok), research/14 §9. **Порядок бэкенду:** **L1 post-check ЛЕММАТИЗИРУЮЩИЙ (pymorphy3), не регэксп** — ПЕРВЫМ (наивный регэксп даёт **18–36% ЛОЖНЫХ флагов** на трудном чанке; pymorphy сворачивает и OOV-транслит Вана→ван; количественная валидация несущего риска research/13 §2). **L2 recall — нормализация ПОЛНАЯ+тестируемая (OpenCC, не мини-карта: живой A4-баг 爺→爷 молча промахнул 赵太爷) + alias-граф + sticky** (recall 0.00→0.50→0.75). **L3 re-ask** (deepseek 0.82→0.91). **L4 таблица Палладия (B6) — потолок ПРАВИЛЬНОСТИ.** Побочно: 钱 (односимвольный гомограф Цянь/деньги) упорно промахивается → **min_key_len/запрет одиночных ключей A3 обязателен.** Поправка §2: soft-глоссарий на лёгком чанке = **~1.0** (0.875 был регэксп-артефакт), demos не бьют — потолок. Артефакты: `eval/adaptive_reask.py`, `eval/adaptive_levers.py`. Инстинкт «тюнить ранкер» — легитимен только во 2-м эшелоне (bge-m3 без разделяющего порога), не в костяке.