package pipeline import ( "encoding/json" "strings" "textmachine/backend/internal/checks" ) // quality.go: the DETERMINISTIC per-run quality-report (D39 layer 5, H5-no-in-loop-quality-signal) — // the online/offline quality telemetry the owner asked for "from day one" (п.25/п.31), which // research/18 §C1 #10 flagged as a NOW lever but was silently deferred to Ф2. It AGGREGATES signals // that are ALREADY computed and stored (retrieval_state: glossary post-check misses, cheap style // flaggers, trust-gated suppressions; chunk_status: echo/CJK/sanitizer flags) plus ONE cheap // deterministic structural KPI (sentences per narrative paragraph — the "choppy paragraphs" claim-1 // signal) recomputed from the exported final text. It is a PURE READ-ONLY projection like Status: $0, // no LLM, no snapshot touch, no checkpoint replay — only OBSERVABILITY, never a gate. The semantic // span-judge (inversion/omission backstop for claim-2) is NOT here — it is research-dependent (pack-2). // QualityReport is the whole-book per-run quality projection. The shipping granularity is the OUTPUT UNIT // (edit unit for an edit pipeline, draft chunk for a draft-only one), so the whole-book counters are named // *_units (naming-debt fix D39.18-follow-up: pre-c-lite they said "chunks" but count units). type QualityReport struct { BookID string `json:"book_id"` TotalUnits int `json:"total_units"` // TextUnits is the number of units whose exported final text was available for the structural KPI // (done or cosmetically-stripped); a flagged-empty unit contributes no prose. TextUnits int `json:"text_units"` // ProcessedUnits is the number of units that REACHED the final stage (a final-stage row exists: ok, // cosmetic-strip, or skipped-because-a-member-flagged) — the strip-rate denominator, so the rate is a // bounded [0,1] fraction of processed units. ProcessedUnits int `json:"processed_units"` // Claim-1 structural KPI (choppy paragraphs). MeanSentPerNarrPara ≈ 1 is choppy (one sentence per // paragraph — the owner's exact complaint); higher is merged discourse prose. Aggregated as // total sentences / total narrative paragraphs across the book, so it is a true book-wide mean. NarrativeSentences int `json:"narrative_sentences"` NarrativeParagraphs int `json:"narrative_paragraphs"` MeanSentPerNarrPara float64 `json:"mean_sentences_per_narrative_paragraph"` // Deterministic signal aggregates (all observability, never a disposition). DialogueDashFlags int `json:"dialogue_dash_flags"` // Rosenthal dialogue-dash inconsistencies GlossaryMisses int `json:"glossary_misses"` // CONFIRMED post-check misses (D10 consistency) NumberDriftFlags int `json:"number_drift_flags"` // reflow number drift + 万/億 magnitude drift TrustGated int `json:"trust_gated"` // lower-trust suppressions refused (seed hygiene) CosmeticStripUnits int `json:"cosmetic_strip_units"` // units the sanitizer auto-stripped (markdown header OR CJK leak — F6) // Echo is SPLIT by stage (owner decision, D39.18-follow-up): a translator echo (draft) and an editor // echo (edit) measure different things and were conflated by the old single echo_rate + a c-lite // re-derive hack. echo_draft = the DRAFT quality (fraction of draft-stage rows flagged cjk_artifact, // INCLUDING a c-lite dropped member — the translator echoed even if the editor recovered the unit), // computed DIRECTLY from the draft rows (no per-unit re-derivation). echo_edit = the DELIVERED quality // (fraction of edit-stage units whose EDITOR output itself echoed — a skipped edit row is a draft echo, // not an editor one, so it is excluded from the numerator). EchoDraftChunks int `json:"echo_draft_chunks"` // draft-stage rows flagged cjk_artifact (per DRAFT chunk) EchoDraftRate float64 `json:"echo_draft_rate"` // over live draft rows EchoEditUnits int `json:"echo_edit_units"` // edit-stage units whose editor output echoed EchoEditRate float64 `json:"echo_edit_rate"` // over live edit rows // CosmeticStripRate is over ProcessedUnits (0..1). It covers BOTH strip classes (a markdown-only strip // is NOT a CJK leak — F6, D39.4: the old cjk_leak_rate counted every sanitizer_stripped unit). CosmeticStripRate float64 `json:"cosmetic_strip_rate"` // RepairCandidates is the ADDRESSABLE-defect residual (pack-16, D39.24): how many defects of the // anchored classes survive in the SHIPPED text of this run, counted after the uniqueness and // disjointness guards — i.e. how many a repair loop would actually attack. It is the measurement that // gates whether the paid loop is worth enabling at all, and it costs $0: the same read-only projection // that already recomputes the structural KPI re-runs the deterministic detectors over the exported text // and the manifest source. Zero for a book whose pair/target ships no checker data. // // CAVEAT (stated, not silent): the scan measures the EXPORT-normalised text, while the in-loop detector // would see the raw completion. The two differ only by the recoverable-glyph fold, so a candidate count // can differ by the rare defect that the export contract itself repairs. RepairCandidates int `json:"repair_candidates,omitempty"` // RepairCandidatesByClass breaks the residual down per class (json.Marshal sorts the keys, so the // rendering is deterministic). nil when nothing fired, so a clean book's report is byte-identical to // what it was before this field existed. RepairCandidatesByClass map[string]int `json:"repair_candidates_by_class,omitempty"` // RepairCalls / RepairApplied / RepairDeclined / RepairRejected are the loop's OUTCOME counters, DERIVED // from durable artifacts rather than stored (§15.2 B): a counter column on retrieval_state would be wiped // by the draft wave's unconditional row rewrite on every resumed run, while checkpoints and the derived // final_hash namespace survive. Declined = the model answered "no change", i.e. OUR flag was the false // positive — the loop's own precision measurement. All omitempty: a book that never repaired is // byte-identical to before these fields existed. RepairCalls int `json:"repair_calls,omitempty"` RepairApplied int `json:"repair_applied,omitempty"` RepairDeclined int `json:"repair_declined,omitempty"` RepairRejected int `json:"repair_rejected,omitempty"` // DegenerateLoopRuns counts runs of ≥ segmentLoopMinRun consecutive units whose exported MODEL text is // byte-identical after whitespace normalization — a degenerate translation loop (pack-13 point-10, // research/21; complements echoMineViolation on the degenerate path research/15). Observability only, // never a disposition or a wire touch; 0 on a healthy book (every unit's source, and so its translation, // differs). The signature is over the MODEL output (checks.ExportNormalize, BEFORE the deterministic title is // prepended), so a per-chapter «Глава N» never masks a body loop. omitempty keeps a loop-free run's // report byte-identical to before this field existed. DegenerateLoopRuns int `json:"degenerate_loop_runs,omitempty"` Chunks []ChunkQuality `json:"chunks,omitempty"` } // ChunkQuality is one chunk's per-chunk quality row (the "where did quality slip" signal). type ChunkQuality struct { Chapter int `json:"chapter"` ChunkIdx int `json:"chunk_idx"` NarrativeSentences int `json:"narrative_sentences"` NarrativeParagraphs int `json:"narrative_paragraphs"` DialogueDashFlags int `json:"dialogue_dash_flags"` GlossaryMisses int `json:"glossary_misses"` NumberDriftFlags int `json:"number_drift_flags"` TrustGated int `json:"trust_gated"` // RepairCandidates is this unit's addressable-defect residual (pack-16); omitted when zero so a clean // unit's row is byte-identical to what it was before the field existed. RepairCandidates int `json:"repair_candidates,omitempty"` } // QualityReport builds the read-only per-run quality projection. It opens no jobs, reserves nothing, // makes no LLM call — it reads the persisted chunk_status / retrieval_state and, for each chunk with // an exported final text, the $0 final checkpoint to recompute the structural KPI. Safe to run // whenever `report`/`status` are (the same exclusive-lock rule). Deterministic over the store. // CAVEAT (same class as Status's config-drift): the exported-text signals key on the CURRENT config's // final-stage NAME; if a config edit renamed the final stage since the run, the stored rows use the // old name and the structural KPI / echo / CJK rates read 0 (the underlying spend/verdict rows are // untouched — surface `status` shows the drift). A run under the same config reads correctly. func (r *Runner) QualityReport() (*QualityReport, error) { statuses, err := r.Store.ChunkStatusesForBook(r.Book.BookID) if err != nil { return nil, err } states, err := r.Store.RetrievalStatesForBook(r.Book.BookID) if err != nil { return nil, err } // Total = the SHIPPING units (the manifest re-chunk, $0), matching status/export + the per-unit // BookResult (R1): under the wave model the editor's final text is per EDIT UNIT, so ProcessedUnits // (the lastStage rows, one per unit leader) and TotalUnits agree at unit granularity. The per-unit // KPI/strip signals land on the leader's edit row; a non-leader member's ChunkQuality row carries // only its draft-side signals (trust-gated) with a 0 structural KPI — observability, never a gate. chunks, err := r.bookChunks() if err != nil { return nil, err } units := r.outputUnits(chunks) // GHOST guard (parity with Export/Status): a stored row whose unit-leader key is NOT in the current // manifest (source shrank since the run) is a ghost — dropping it keeps ProcessedUnits ≤ TotalUnits. inManifest := map[chunkKey]bool{} for _, u := range units { inManifest[chunkKey{u.Chapter, u.FirstChunkIdx}] = true } // A retrieval_state row / a draft echo is keyed per DRAFT chunk, so it is live iff its chunk is still in // the manifest; a chunk that left the source is a ghost. liveChunks := map[chunkKey]bool{} for _, ch := range chunks { liveChunks[chunkKey{ch.Chapter, ch.ChunkIdx}] = true } // Unit source, keyed by the leader row the final-stage verdict lives on — the src side the comparative // detectors need. It is the $0 manifest re-chunk (never a stored or billed artifact), joined exactly as // the editor's input is (wave.go sourceText). unitSource := make(map[chunkKey]string, len(units)) for _, u := range units { unitSource[chunkKey{u.Chapter, u.FirstChunkIdx}] = u.sourceText() } repairCfg := r.cheapGateConfig() // Same target gate the in-loop path applies: the Latin-residue class cannot be made inert by data, so on // a Latin-script target it would report every word as an addressable defect and make the residual // meaningless. A target without the readability gate simply reports the data-driven classes. repairLatinOK := isRuTarget(r.Book.TargetLang) // Wave stage-name sets classify a stored chunk_status row by wave for the split echo metric (D39.18 // owner decision): echo_draft is counted DIRECTLY from the draft rows (no c-lite per-unit re-derivation // — a dropped member's own draft row already carries cjk_artifact), echo_edit from the edit rows. draftStageNames := stageNameSet(r.waveStagesIndexed(waveDraft)) editStageNames := stageNameSet(r.waveStagesIndexed(waveEdit)) rep := &QualityReport{BookID: r.Book.BookID, TotalUnits: len(units)} byChunk := map[chunkKey]*ChunkQuality{} order := []chunkKey{} chunkOf := func(k chunkKey) *ChunkQuality { if q := byChunk[k]; q != nil { return q } q := &ChunkQuality{Chapter: k.chapter, ChunkIdx: k.chunkIdx} byChunk[k] = q order = append(order, k) return q } // The glossary post-check GATE flips a chunk to withheld (flagged glossary_miss, empty export) at // the CHUNK level without a chunk_status row (like Export/Status re-derive it). F5 (D39.4): the // structural KPI must EXCLUDE these — their export is "", so counting their (withheld) text in // TextUnits/KPI diverges from what `tmctl export` and `translate` actually ship. gateOn := r.Pipeline.Gates.Glossary.PostcheckGate withheld := map[chunkKey]bool{} // Aggregate the stored retrieval-state signals (glossary consistency, style breakdown, trust-gated). // A retrieval_state row is keyed per DRAFT chunk (live iff its chunk is still in the manifest); a leader // row also carries the unit's post-check, so if the leader survives, so does the unit. for _, rs := range states { if !liveChunks[chunkKey{rs.Chapter, rs.ChunkIdx}] { continue // ghost retrieval_state row (chunk dropped from source) } q := chunkOf(chunkKey{rs.Chapter, rs.ChunkIdx}) q.GlossaryMisses += rs.NPostcheckMiss q.TrustGated += rs.NTrustGatedSuppress rep.GlossaryMisses += rs.NPostcheckMiss rep.TrustGated += rs.NTrustGatedSuppress if gateOn && rs.NPostcheckMiss > 0 { withheld[chunkKey{rs.Chapter, rs.ChunkIdx}] = true } if rs.NStyleFlags > 0 && rs.StyleDetail != "" { var cg checks.CheapGateResult if json.Unmarshal([]byte(rs.StyleDetail), &cg) == nil { dash := cg.DialogueDash drift := cg.NumberDrift + cg.NumberMagnitude q.DialogueDashFlags += dash q.NumberDriftFlags += drift rep.DialogueDashFlags += dash rep.NumberDriftFlags += drift } } } // Split echo by stage (D39.18 owner decision): echo_draft over the DRAFT-stage rows (translator quality, // INCLUDING a c-lite dropped member — its own draft row carries cjk_artifact, so no per-unit re-derivation // is needed), echo_edit over the EDIT-stage rows (delivered quality — only the EDITOR's own echo counts, // a skipped edit row is a draft echo not an editor one). Ghost-guarded like the rest (live chunks / units). var draftRows, draftEcho, editRows, editEcho int for _, cs := range statuses { k := chunkKey{cs.Chapter, cs.ChunkIdx} switch { case draftStageNames[cs.Stage] && liveChunks[k]: draftRows++ if cs.FlagReason == string(FlagCJKArtifact) { draftEcho++ // the translator echoed CJK (a flagged draft row), incl. a c-lite dropped member } case editStageNames[cs.Stage] && inManifest[k]: editRows++ if cs.Disposition == string(DispFlagged) && cs.FlagReason == string(FlagCJKArtifact) { editEcho++ // the EDITOR's OWN output echoed (a skipped edit row means the drafts echoed, not the editor) } } } rep.EchoDraftChunks, rep.EchoEditUnits = draftEcho, editEcho if draftRows > 0 { rep.EchoDraftRate = float64(draftEcho) / float64(draftRows) } if editRows > 0 { rep.EchoEditRate = float64(editEcho) / float64(editRows) } // The exported-text units (the FINAL stage's row): the cosmetic-strip rate + the structural KPI. lastStage := "" if n := len(r.Pipeline.Stages); n > 0 { lastStage = r.Pipeline.Stages[n-1].Name } loopText := map[chunkKey]string{} // per-unit exported MODEL text (no title), for the pack-13 point-10 loop scan for _, cs := range statuses { k := chunkKey{cs.Chapter, cs.ChunkIdx} if cs.Stage != lastStage || !inManifest[k] { continue // the final verdict lives on the final stage's row (per unit); drop ghost leader rows } // Every unit that REACHED the final stage has exactly one lastStage row (ok, cosmetic-strip, or // skipped-because-a-member-flagged) — the strip-rate denominator. rep.ProcessedUnits++ if cs.FlagReason == string(FlagSanitizerStripped) { rep.CosmeticStripUnits++ // a stripped unit carried a markdown OR CJK cosmetic leak (F6) } // Structural KPI: recompute over the exported final text (ok, or the cosmetic-stripped export). if cs.FinalHash == "" { continue } if cs.Disposition != string(DispOK) && cs.FlagReason != string(FlagSanitizerStripped) { continue // a dropped unit exported nothing } if withheld[k] { continue // F5: the glossary gate withheld this unit's text — it exports nothing } cp, cperr := r.Store.GetCheckpoint(cs.FinalHash) if cperr != nil { return nil, cperr } if cp == nil || strings.TrimSpace(cp.ResponseText) == "" { continue } normText := checks.ExportNormalize(cp.ResponseText) loopText[k] = normText // Addressable-defect residual (pack-16): the same guards the repair sub-step applies — uniqueness // inside RepairCandidates, then sentence-expansion + disjointness — so the number is «how many // repairs would actually be attempted», not «how many flags exist». scanned := checks.RepairCandidates(unitSource[k], normText, repairCfg) if !repairLatinOK { kept := scanned[:0] for _, c := range scanned { if c.Class != checks.RepairLatinResidue { kept = append(kept, c) } } scanned = kept } if cands := checks.DisjointCandidates(normText, scanned); len(cands) > 0 { if rep.RepairCandidatesByClass == nil { rep.RepairCandidatesByClass = map[string]int{} } for _, c := range cands { rep.RepairCandidatesByClass[string(c.Class)]++ } rep.RepairCandidates += len(cands) chunkOf(k).RepairCandidates += len(cands) } sent, para := checks.NarrativeStructure(normText) q := chunkOf(k) q.NarrativeSentences, q.NarrativeParagraphs = sent, para rep.NarrativeSentences += sent rep.NarrativeParagraphs += para rep.TextUnits++ } // Degenerate-loop guard (pack-13 point-10): scan the exported MODEL texts in reading (manifest/unit) // order for runs of identical consecutive units — observability, never a gate. Deterministic. ordered := make([]string, 0, len(units)) for _, u := range units { ordered = append(ordered, loopText[chunkKey{u.Chapter, u.FirstChunkIdx}]) } rep.DegenerateLoopRuns = len(segmentLoopRuns(ordered, segmentLoopMinRun)) // Repair outcomes, derived (never stored) — see the field docs. if calls, declined, applied, rerr := r.Store.RepairStats(r.Book.BookID, repairNoChange, repairDerivedNS+":"); rerr == nil { rep.RepairCalls, rep.RepairDeclined, rep.RepairApplied = calls, declined, applied if n := calls - declined - applied; n > 0 { rep.RepairRejected = n // paid, neither declined nor applied ⇒ a guard or the re-gate refused it } } else { return nil, rerr } if rep.NarrativeParagraphs > 0 { rep.MeanSentPerNarrPara = float64(rep.NarrativeSentences) / float64(rep.NarrativeParagraphs) } if rep.ProcessedUnits > 0 { rep.CosmeticStripRate = float64(rep.CosmeticStripUnits) / float64(rep.ProcessedUnits) } for _, k := range order { rep.Chunks = append(rep.Chunks, *byChunk[k]) } return rep, nil }