textmachine/backend/internal/pipeline/quality.go

269 lines
13 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

package pipeline
import (
"encoding/json"
"strings"
)
// quality.go: the DETERMINISTIC per-run quality-report (D39 layer 5, H5-no-in-loop-quality-signal) —
// the online/offline quality telemetry the owner asked for "from day one" (п.25/п.31), which
// research/18 §C1 #10 flagged as a NOW lever but was silently deferred to Ф2. It AGGREGATES signals
// that are ALREADY computed and stored (retrieval_state: glossary post-check misses, cheap style
// flaggers, trust-gated suppressions; chunk_status: echo/CJK/sanitizer flags) plus ONE cheap
// deterministic structural KPI (sentences per narrative paragraph — the "choppy paragraphs" claim-1
// signal) recomputed from the exported final text. It is a PURE READ-ONLY projection like Status: $0,
// no LLM, no snapshot touch, no checkpoint replay — only OBSERVABILITY, never a gate. The semantic
// span-judge (inversion/omission backstop for claim-2) is NOT here — it is research-dependent (pack-2).
// QualityReport is the whole-book per-run quality projection. The shipping granularity is the OUTPUT UNIT
// (edit unit for an edit pipeline, draft chunk for a draft-only one), so the whole-book counters are named
// *_units (naming-debt fix D39.18-follow-up: pre-c-lite they said "chunks" but count units).
type QualityReport struct {
BookID string `json:"book_id"`
TotalUnits int `json:"total_units"`
// TextUnits is the number of units whose exported final text was available for the structural KPI
// (done or cosmetically-stripped); a flagged-empty unit contributes no prose.
TextUnits int `json:"text_units"`
// ProcessedUnits is the number of units that REACHED the final stage (a final-stage row exists: ok,
// cosmetic-strip, or skipped-because-a-member-flagged) — the strip-rate denominator, so the rate is a
// bounded [0,1] fraction of processed units.
ProcessedUnits int `json:"processed_units"`
// Claim-1 structural KPI (choppy paragraphs). MeanSentPerNarrPara ≈ 1 is choppy (one sentence per
// paragraph — the owner's exact complaint); higher is merged discourse prose. Aggregated as
// total sentences / total narrative paragraphs across the book, so it is a true book-wide mean.
NarrativeSentences int `json:"narrative_sentences"`
NarrativeParagraphs int `json:"narrative_paragraphs"`
MeanSentPerNarrPara float64 `json:"mean_sentences_per_narrative_paragraph"`
// Deterministic signal aggregates (all observability, never a disposition).
DialogueDashFlags int `json:"dialogue_dash_flags"` // Rosenthal dialogue-dash inconsistencies
GlossaryMisses int `json:"glossary_misses"` // CONFIRMED post-check misses (D10 consistency)
NumberDriftFlags int `json:"number_drift_flags"` // reflow number drift + 万/億 magnitude drift
TrustGated int `json:"trust_gated"` // lower-trust suppressions refused (seed hygiene)
CosmeticStripUnits int `json:"cosmetic_strip_units"` // units the sanitizer auto-stripped (markdown header OR CJK leak — F6)
// Echo is SPLIT by stage (owner decision, D39.18-follow-up): a translator echo (draft) and an editor
// echo (edit) measure different things and were conflated by the old single echo_rate + a c-lite
// re-derive hack. echo_draft = the DRAFT quality (fraction of draft-stage rows flagged cjk_artifact,
// INCLUDING a c-lite dropped member — the translator echoed even if the editor recovered the unit),
// computed DIRECTLY from the draft rows (no per-unit re-derivation). echo_edit = the DELIVERED quality
// (fraction of edit-stage units whose EDITOR output itself echoed — a skipped edit row is a draft echo,
// not an editor one, so it is excluded from the numerator).
EchoDraftChunks int `json:"echo_draft_chunks"` // draft-stage rows flagged cjk_artifact (per DRAFT chunk)
EchoDraftRate float64 `json:"echo_draft_rate"` // over live draft rows
EchoEditUnits int `json:"echo_edit_units"` // edit-stage units whose editor output echoed
EchoEditRate float64 `json:"echo_edit_rate"` // over live edit rows
// CosmeticStripRate is over ProcessedUnits (0..1). It covers BOTH strip classes (a markdown-only strip
// is NOT a CJK leak — F6, D39.4: the old cjk_leak_rate counted every sanitizer_stripped unit).
CosmeticStripRate float64 `json:"cosmetic_strip_rate"`
Chunks []ChunkQuality `json:"chunks,omitempty"`
}
// ChunkQuality is one chunk's per-chunk quality row (the "where did quality slip" signal).
type ChunkQuality struct {
Chapter int `json:"chapter"`
ChunkIdx int `json:"chunk_idx"`
NarrativeSentences int `json:"narrative_sentences"`
NarrativeParagraphs int `json:"narrative_paragraphs"`
DialogueDashFlags int `json:"dialogue_dash_flags"`
GlossaryMisses int `json:"glossary_misses"`
NumberDriftFlags int `json:"number_drift_flags"`
TrustGated int `json:"trust_gated"`
}
// narrativeStructure computes the claim-1 structural KPI over a FINAL Russian text: the count of
// NARRATIVE paragraphs (non-empty lines that are NOT a dialogue turn opened with a dash «—»/«–»/«-»)
// and the total sentences within them (the oracle-parity splitSentences). Dialogue turns are excluded
// (they are legitimately one short line). Deterministic and pure — the same signal the reflow lever
// is supposed to move (exp14 mean-sentences-per-paragraph), reused here as in-loop observability.
func narrativeStructure(final string) (sentences, paragraphs int) {
for _, line := range strings.Split(final, "\n") {
t := strings.TrimSpace(line)
if t == "" {
continue
}
if rs := []rune(t); rs[0] == '—' || rs[0] == '' || rs[0] == '-' {
continue // a dialogue turn — not a narrative paragraph
}
paragraphs++
sentences += len(splitSentences(t))
}
return sentences, paragraphs
}
// QualityReport builds the read-only per-run quality projection. It opens no jobs, reserves nothing,
// makes no LLM call — it reads the persisted chunk_status / retrieval_state and, for each chunk with
// an exported final text, the $0 final checkpoint to recompute the structural KPI. Safe to run
// whenever `report`/`status` are (the same exclusive-lock rule). Deterministic over the store.
// CAVEAT (same class as Status's config-drift): the exported-text signals key on the CURRENT config's
// final-stage NAME; if a config edit renamed the final stage since the run, the stored rows use the
// old name and the structural KPI / echo / CJK rates read 0 (the underlying spend/verdict rows are
// untouched — surface `status` shows the drift). A run under the same config reads correctly.
func (r *Runner) QualityReport() (*QualityReport, error) {
statuses, err := r.Store.ChunkStatusesForBook(r.Book.BookID)
if err != nil {
return nil, err
}
states, err := r.Store.RetrievalStatesForBook(r.Book.BookID)
if err != nil {
return nil, err
}
// Total = the SHIPPING units (the manifest re-chunk, $0), matching status/export + the per-unit
// BookResult (R1): under the wave model the editor's final text is per EDIT UNIT, so ProcessedUnits
// (the lastStage rows, one per unit leader) and TotalUnits agree at unit granularity. The per-unit
// KPI/strip signals land on the leader's edit row; a non-leader member's ChunkQuality row carries
// only its draft-side signals (trust-gated) with a 0 structural KPI — observability, never a gate.
chunks, err := r.bookChunks()
if err != nil {
return nil, err
}
units := r.outputUnits(chunks)
// GHOST guard (parity with Export/Status): a stored row whose unit-leader key is NOT in the current
// manifest (source shrank since the run) is a ghost — dropping it keeps ProcessedUnits ≤ TotalUnits.
inManifest := map[chunkKey]bool{}
for _, u := range units {
inManifest[chunkKey{u.Chapter, u.FirstChunkIdx}] = true
}
// A retrieval_state row / a draft echo is keyed per DRAFT chunk, so it is live iff its chunk is still in
// the manifest; a chunk that left the source is a ghost.
liveChunks := map[chunkKey]bool{}
for _, ch := range chunks {
liveChunks[chunkKey{ch.Chapter, ch.ChunkIdx}] = true
}
// Wave stage-name sets classify a stored chunk_status row by wave for the split echo metric (D39.18
// owner decision): echo_draft is counted DIRECTLY from the draft rows (no c-lite per-unit re-derivation
// — a dropped member's own draft row already carries cjk_artifact), echo_edit from the edit rows.
draftStageNames := stageNameSet(r.waveStagesIndexed(waveDraft))
editStageNames := stageNameSet(r.waveStagesIndexed(waveEdit))
rep := &QualityReport{BookID: r.Book.BookID, TotalUnits: len(units)}
byChunk := map[chunkKey]*ChunkQuality{}
order := []chunkKey{}
chunkOf := func(k chunkKey) *ChunkQuality {
if q := byChunk[k]; q != nil {
return q
}
q := &ChunkQuality{Chapter: k.chapter, ChunkIdx: k.chunkIdx}
byChunk[k] = q
order = append(order, k)
return q
}
// The glossary post-check GATE flips a chunk to withheld (flagged glossary_miss, empty export) at
// the CHUNK level without a chunk_status row (like Export/Status re-derive it). F5 (D39.4): the
// structural KPI must EXCLUDE these — their export is "", so counting their (withheld) text in
// TextUnits/KPI diverges from what `tmctl export` and `translate` actually ship.
gateOn := r.Pipeline.Gates.Glossary.PostcheckGate
withheld := map[chunkKey]bool{}
// Aggregate the stored retrieval-state signals (glossary consistency, style breakdown, trust-gated).
// A retrieval_state row is keyed per DRAFT chunk (live iff its chunk is still in the manifest); a leader
// row also carries the unit's post-check, so if the leader survives, so does the unit.
for _, rs := range states {
if !liveChunks[chunkKey{rs.Chapter, rs.ChunkIdx}] {
continue // ghost retrieval_state row (chunk dropped from source)
}
q := chunkOf(chunkKey{rs.Chapter, rs.ChunkIdx})
q.GlossaryMisses += rs.NPostcheckMiss
q.TrustGated += rs.NTrustGatedSuppress
rep.GlossaryMisses += rs.NPostcheckMiss
rep.TrustGated += rs.NTrustGatedSuppress
if gateOn && rs.NPostcheckMiss > 0 {
withheld[chunkKey{rs.Chapter, rs.ChunkIdx}] = true
}
if rs.NStyleFlags > 0 && rs.StyleDetail != "" {
var cg cheapGateResult
if json.Unmarshal([]byte(rs.StyleDetail), &cg) == nil {
dash := cg.DialogueDash
drift := cg.NumberDrift + cg.NumberMagnitude
q.DialogueDashFlags += dash
q.NumberDriftFlags += drift
rep.DialogueDashFlags += dash
rep.NumberDriftFlags += drift
}
}
}
// Split echo by stage (D39.18 owner decision): echo_draft over the DRAFT-stage rows (translator quality,
// INCLUDING a c-lite dropped member — its own draft row carries cjk_artifact, so no per-unit re-derivation
// is needed), echo_edit over the EDIT-stage rows (delivered quality — only the EDITOR's own echo counts,
// a skipped edit row is a draft echo not an editor one). Ghost-guarded like the rest (live chunks / units).
var draftRows, draftEcho, editRows, editEcho int
for _, cs := range statuses {
k := chunkKey{cs.Chapter, cs.ChunkIdx}
switch {
case draftStageNames[cs.Stage] && liveChunks[k]:
draftRows++
if cs.FlagReason == string(FlagCJKArtifact) {
draftEcho++ // the translator echoed CJK (a flagged draft row), incl. a c-lite dropped member
}
case editStageNames[cs.Stage] && inManifest[k]:
editRows++
if cs.Disposition == string(DispFlagged) && cs.FlagReason == string(FlagCJKArtifact) {
editEcho++ // the EDITOR's OWN output echoed (a skipped edit row means the drafts echoed, not the editor)
}
}
}
rep.EchoDraftChunks, rep.EchoEditUnits = draftEcho, editEcho
if draftRows > 0 {
rep.EchoDraftRate = float64(draftEcho) / float64(draftRows)
}
if editRows > 0 {
rep.EchoEditRate = float64(editEcho) / float64(editRows)
}
// The exported-text units (the FINAL stage's row): the cosmetic-strip rate + the structural KPI.
lastStage := ""
if n := len(r.Pipeline.Stages); n > 0 {
lastStage = r.Pipeline.Stages[n-1].Name
}
for _, cs := range statuses {
k := chunkKey{cs.Chapter, cs.ChunkIdx}
if cs.Stage != lastStage || !inManifest[k] {
continue // the final verdict lives on the final stage's row (per unit); drop ghost leader rows
}
// Every unit that REACHED the final stage has exactly one lastStage row (ok, cosmetic-strip, or
// skipped-because-a-member-flagged) — the strip-rate denominator.
rep.ProcessedUnits++
if cs.FlagReason == string(FlagSanitizerStripped) {
rep.CosmeticStripUnits++ // a stripped unit carried a markdown OR CJK cosmetic leak (F6)
}
// Structural KPI: recompute over the exported final text (ok, or the cosmetic-stripped export).
if cs.FinalHash == "" {
continue
}
if cs.Disposition != string(DispOK) && cs.FlagReason != string(FlagSanitizerStripped) {
continue // a dropped unit exported nothing
}
if withheld[k] {
continue // F5: the glossary gate withheld this unit's text — it exports nothing
}
cp, cperr := r.Store.GetCheckpoint(cs.FinalHash)
if cperr != nil {
return nil, cperr
}
if cp == nil || strings.TrimSpace(cp.ResponseText) == "" {
continue
}
sent, para := narrativeStructure(exportNormalize(cp.ResponseText))
q := chunkOf(k)
q.NarrativeSentences, q.NarrativeParagraphs = sent, para
rep.NarrativeSentences += sent
rep.NarrativeParagraphs += para
rep.TextUnits++
}
if rep.NarrativeParagraphs > 0 {
rep.MeanSentPerNarrPara = float64(rep.NarrativeSentences) / float64(rep.NarrativeParagraphs)
}
if rep.ProcessedUnits > 0 {
rep.CosmeticStripRate = float64(rep.CosmeticStripUnits) / float64(rep.ProcessedUnits)
}
for _, k := range order {
rep.Chunks = append(rep.Chunks, *byChunk[k])
}
return rep, nil
}