package pipeline
import (
"strings"
"testing"
"textmachine/backend/internal/config"
)
// coverage_test.go pins the excision coverage-gate. The segmentation/metric is
// PINNED BIT-FOR-BIT to the Polygon acceptance oracle (eval/refusal_bench.py,
// D12 Q1): the expected values below were captured from the LIVE oracle
// (scripts/oracle_groundtruth.py over refusal_bench.split_sentences/cjk_share). If
// the oracle changes, this table must change Python-first, in lock-step — that is
// the contract, not an accident.
// oracleSplit is the ground-truth from refusal_bench.py::split_sentences.
var oracleSplit = []struct {
in string
want []string
}{
{"", nil},
{" ", nil},
{"Hello world", []string{"Hello world"}},
{"Hello. World.", []string{"Hello.", "World."}},
{"Hello.World.", []string{"Hello.World."}}, // no space after Latin '.' → no split
{"Hello!! World?? Yes.", []string{"Hello!!", "World??", "Yes."}},
{"One. Two! Three? Four…", []string{"One.", "Two!", "Three?", "Four…"}},
{"Wait… really?", []string{"Wait…", "really?"}},
{"猫。", []string{"猫。"}},
{"猫。狗。鸟。", []string{"猫。", "狗。", "鸟。"}},
{"猫。 狗。 鸟。", []string{"猫。", "狗。", "鸟。"}},
{"猫。。狗", []string{"猫。", "。", "狗"}}, // consecutive CJK terminators → lone 。 fragment
{"「こんにちは!」彼は笑った。", []string{"「こんにちは!", "」彼は笑った。"}}, // ! splits zero-width, 」 joins next
{"『引用』。次の文。", []string{"『引用』。", "次の文。"}},
{"«Привет!» — он засмеялся. «Пока».", []string{"«Привет!» — он засмеялся.", "«Пока»."}}, // !» does NOT split
{"“Hi!” he said. “Bye.”", []string{"“Hi!” he said.", "“Bye.”"}},
{"阿Q正传。第一章。", []string{"阿Q正传。", "第一章。"}},
{"彼は言った:「行こう!」", []string{"彼は言った:「行こう!", "」"}},
{"はい。 いいえ。", []string{"はい。", "いいえ。"}}, // U+3000 ideographic space after 。 is consumed
{"Yes.\tNo.\nMaybe.", []string{"Yes.", "No.", "Maybe."}},
{"こんにちは、世界。さようなら!", []string{"こんにちは、世界。", "さようなら!"}}, // 、 is NOT a terminator
{"……", []string{"……"}}, // Latin ellipsis, no whitespace → no split
{"。。。", []string{"。", "。", "。"}}, // each CJK period splits zero-width
{"A。B. C。D", []string{"A。", "B.", "C。", "D"}},
{"Full?width。terminators!here", []string{"Full?", "width。", "terminators!", "here"}},
{" leading and trailing. ", []string{"leading and trailing."}},
{"第一。\n\n第二。", []string{"第一。", "第二。"}},
{"Mr. Smith went. He left.", []string{"Mr.", "Smith went.", "He left."}}, // naive over-split (known)
// U+001F (unit separator) is whitespace to Python (\s / str.isspace) but NOT to
// Go's unicode.IsSpace — isOracleSpace bridges the gap for oracle parity.
{"A.B.", []string{"A.", "B."}},
}
func TestSplitSentencesOracleParity(t *testing.T) {
for _, c := range oracleSplit {
got := splitSentences(c.in)
if len(got) != len(c.want) {
t.Errorf("splitSentences(%q) = %#v (n=%d), want %#v (n=%d)", c.in, got, len(got), c.want, len(c.want))
continue
}
for i := range got {
if got[i] != c.want[i] {
t.Errorf("splitSentences(%q)[%d] = %q, want %q", c.in, i, got[i], c.want[i])
}
}
}
}
func TestNonSpaceCountOracleParity(t *testing.T) {
cases := []struct {
in string
want int
}{
{"Hello. World.", 12},
{"猫。 狗。 鸟。", 6},
{"Yes.\tNo.\nMaybe.", 13},
{"はい。 いいえ。", 7}, // U+3000 counts as space → excluded
{" leading and trailing. ", 19},
{"", 0},
}
for _, c := range cases {
if got := nonSpaceCount(c.in); got != c.want {
t.Errorf("nonSpaceCount(%q) = %d, want %d", c.in, got, c.want)
}
}
}
// boevoyBounds mirrors the boevoy pipeline-c1.yaml corridors.
func boevoyGate(minChars int) config.CoverageGate {
return config.CoverageGate{
Enabled: true,
LenRatio: map[string][]float64{"zh-ru": {2.2, 4.2}, "ja-ru": {1.4, 2.6}, "en-ru": {0.7, 1.4}},
SentCovMin: 0.75,
MinChunkChars: minChars,
}
}
// TestOracleSpaceControlChars pins isOracleSpace to the Python oracle on the C0
// information separators U+001C–U+001F, which Python treats as whitespace (\s /
// str.isspace) but Go's unicode.IsSpace does not — a real bit-parity gap (self-review).
func TestOracleSpaceControlChars(t *testing.T) {
us := "" // U+001F unit separator (survives EPUB/PDF extraction)
if got := splitSentences("A." + us + "B."); len(got) != 2 || got[0] != "A." || got[1] != "B." {
t.Fatalf("U+001F after a terminator must split like the oracle, got %#v", got)
}
if got := nonSpaceCount("A" + us + "B"); got != 2 {
t.Fatalf("U+001F must count as whitespace (oracle parity), got %d", got)
}
}
// TestCoverageCheckStripsThink guards that a leaked … block is stripped
// BEFORE measuring (parity with classify() and the oracle), so a reasoning block does
// not inflate the counts and MASK a real excision (self-review finding). Mutation:
// drop the stripThink in coverageCheck and this goes green-to-red.
func TestCoverageCheckStripsThink(t *testing.T) {
src := strings.Repeat("这是一个相当长的测试句子。", 20) // 20 sentences
leaked := "" + strings.Repeat("Долгое рассуждение о выборе слова. ", 40) + "Короткий огрызок."
res := coverageCheck(boevoyGate(0), src, leaked, "zh", "ru")
if res.cls.Reason != FlagExcisionSuspect {
t.Fatalf("a block must be stripped so the excision behind it is still caught, got %+v", res)
}
}
func TestCoverageCheckExcision(t *testing.T) {
// Mirrors the three classify_output cases captured from the live oracle.
zhExcised := coverageCheck(boevoyGate(0), strings.Repeat("这是一个测试。", 20), "Короткий перевод.", "zh", "ru")
if zhExcised.cls.Reason != FlagExcisionSuspect {
t.Fatalf("zh short output must flag excision, got %+v", zhExcised)
}
if zhExcised.sentCov > 0.1 || zhExcised.lenRatio > 0.2 {
t.Fatalf("zh metrics off (oracle: sent_cov=0.05, len_ratio=0.114): %+v", zhExcised)
}
zhHealthy := coverageCheck(boevoyGate(0),
strings.Repeat("这是第一句。这是第二句。这是第三句。", 4),
strings.Repeat("Это первое предложение. Это второе предложение. Это третье предложение. ", 4), "zh", "ru")
if !zhHealthy.cls.ok() {
t.Fatalf("zh healthy ~3.5× translation must pass, got %+v", zhHealthy)
}
// ja: sentence coverage 0.25 flags even though len_ratio (1.52) is inside [1.4,2.6].
jaExcised := coverageCheck(boevoyGate(0),
"最初の文。二番目の文。三番目の文。四番目の文。", "Только одно предложение вместо четырёх.", "ja", "ru")
if jaExcised.cls.Reason != FlagExcisionSuspect || !strings.Contains(jaExcised.cls.Detail, "sent_cov") {
t.Fatalf("ja must flag on sentence coverage alone, got %+v", jaExcised)
}
if strings.Contains(jaExcised.cls.Detail, "len_ratio") {
t.Fatalf("ja len_ratio 1.52 is inside the corridor — must NOT be in the detail: %q", jaExcised.cls.Detail)
}
}
// TestCoverageDefaultCorridorForUnknownPair pins external-review parity: a pair absent
// from len_ratio_bounds still gets the ORACLE's default corridor (0.5), not a silently
// skipped len check. sent_cov is kept at 1.0 so ONLY the default len_ratio can flag.
// Mutation: revert the default corridor and this goes green-to-red.
func TestCoverageDefaultCorridorForUnknownPair(t *testing.T) {
gate := config.CoverageGate{Enabled: true, SentCovMin: 0.75, MinChunkChars: 0} // no LenRatio map at all
src := strings.Repeat("가나다라마바사아자차카타파。", 3) // 3 ko sentences, long
out := "다. 라. 마." // 3 sentences (sent_cov 1.0) but tiny → len_ratio ≪ 0.5
res := coverageCheck(gate, src, out, "ko", "ru")
if res.cls.Reason != FlagExcisionSuspect || !strings.Contains(res.cls.Detail, "len_ratio") {
t.Fatalf("a no-corridor pair must apply the oracle default (0.5) len_ratio, got %+v", res)
}
if strings.Contains(res.cls.Detail, "sent_cov") {
t.Fatalf("sent_cov 1.0 must not flag here — the len_ratio default corridor is what fires: %q", res.cls.Detail)
}
}
func TestCoverageCheckApplicabilityAndCorridor(t *testing.T) {
// Below min_chunk_chars → gate is NOT applied (short-chunk ratios are noise).
short := coverageCheck(boevoyGate(500), "这是一个测试。", "x", "zh", "ru")
if short.applied || !short.cls.ok() {
t.Fatalf("a chunk shorter than min_chunk_chars must be skipped, got %+v", short)
}
// No corridor for the pair → len_ratio is skipped but sentence coverage still
// applies (never silently pass everything).
noCorridor := coverageCheck(boevoyGate(0),
"가나。다라。마바。사아。", "Одно предложение.", "ko", "ru")
if noCorridor.cls.Reason != FlagExcisionSuspect || strings.Contains(noCorridor.cls.Detail, "len_ratio") {
t.Fatalf("no-corridor pair must still flag on sent_cov (no len_ratio), got %+v", noCorridor)
}
// Upper-bound anomaly (len_ratio > hi) is DELIBERATELY NOT flagged (D12 Q3):
// matching sentence count (sent_cov=1.0) but a very long per-sentence output so
// len_ratio ≫ 4.2. The gate must still pass — additions are Phase-2 territory.
over := coverageCheck(boevoyGate(0),
strings.Repeat("这是句子。", 10),
strings.Repeat("Это очень длинное предложение с большим количеством слов. ", 10), "zh", "ru")
if !over.cls.ok() {
t.Fatalf("an over-length output must NOT flag (upper bound is out of scope, D12 Q3), got %+v", over)
}
if over.lenRatio <= 4.2 {
t.Fatalf("test premise broken: len_ratio must exceed the upper bound to prove it is ignored, got %.2f", over.lenRatio)
}
}
// TestCoverageGateCatchesArtificialExcision is the Phase-1 acceptance shape (§A):
// a mini-set of fragments with sentences excised from the translation — the gate
// must catch ≥90%. Built synthetically by dropping a fraction of the output
// sentences from an otherwise-faithful translation; the control set (no drop) must
// produce ZERO false flags (precision). Honest scope: this proves the MECHANISM at
// the calibrated thresholds; real-corpus ≥90% is the Polygon oracle + owner precision
// validation (D12 Q4).
func TestCoverageGateCatchesArtificialExcision(t *testing.T) {
gate := boevoyGate(0)
// (srcLang, source sentences, faithful ru sentences) triples, ≥ ~40 sentences
// each so a fractional drop is meaningful.
build := func(unit string, n int) string { return strings.Repeat(unit, n) }
fixtures := []struct {
lang string
src string
full string
}{
{"zh", build("这是一个相当长的句子用于测试覆盖率。", 40), build("Это довольно длинное предложение для проверки покрытия. ", 40)},
{"ja", build("これは覆盖率を検証するための長い文である。", 40), build("Это длинное предложение для проверки покрытия текста. ", 40)},
{"en", build("This is a reasonably long sentence used to test coverage. ", 40), build("Это разумно длинное предложение для проверки покрытия. ", 40)},
}
dropFractions := []float64{0.30, 0.40, 0.50, 0.60} // >25% → sent_cov < 0.75
caught, total := 0, 0
for _, f := range fixtures {
outSents := splitSentences(f.full)
for _, frac := range dropFractions {
keep := int(float64(len(outSents)) * (1 - frac))
excised := strings.Join(outSents[:keep], " ")
res := coverageCheck(gate, f.src, excised, f.lang, "ru")
total++
if res.cls.Reason == FlagExcisionSuspect {
caught++
} else {
t.Logf("MISS: %s drop=%.0f%% sent_cov=%.2f len_ratio=%.2f", f.lang, frac*100, res.sentCov, res.lenRatio)
}
}
// Control: the faithful full translation must NOT flag (precision).
if res := coverageCheck(gate, f.src, f.full, f.lang, "ru"); res.cls.Reason == FlagExcisionSuspect {
t.Errorf("FALSE POSITIVE on faithful %s translation: %s (sent_cov=%.2f len_ratio=%.2f)", f.lang, res.cls.Detail, res.sentCov, res.lenRatio)
}
}
rate := float64(caught) / float64(total)
if rate < 0.90 {
t.Fatalf("coverage gate caught %d/%d (%.0f%%) artificial excisions — below the ≥90%% acceptance bar", caught, total, rate*100)
}
t.Logf("mini-set: caught %d/%d (%.0f%%) artificial excisions", caught, total, rate*100)
}