package pipeline import ( "regexp" "strings" "unicode" ) // banknote.go: the banknote-v1 in-band footnote channel (WS4, план §4 / research/20 §B3 / exp16 // banknote.py — RATIFIED D39.10 as the dst-delivery channel). The translator, AFTER the translation, // MAY emit a versioned separator + tab-delimited term lines for NEW terms only (not in the injected // glossary). This file is the pure, deterministic PARSER + SLICER — the wire-invariant core; the // integration seams (slice BEFORE classifyOutput, derived checkpoint, snapshot fold, telemetry — // план §4(а) 12 points) consume these functions. A faithful Go↔Python port (banknote.py is the // reference): same separator, same tolerant field split, same Han-in-src rule, same truncation // tolerance. Pin (ws4_banknote_verify.py): re-parse of 95 saved lines = 0 parse_fail / 0 truncated. // bankSeparator marks the start of the banknote block. Chosen to (a) not occur in natural prose and // (b) NOT be a "Примечание/Сноска/Комментарий" trailing-note word the sanitizer reserves (§B3-1), so // the two channels never collide. Changing it is a wire/parser change → a loud --resnapshot (§4в). const bankSeparator = "⟦TM-BANK-v1⟧" // bankMaxLines is the per-chunk banknote line budget (§B3-5, feeds the max_tokens sizing at // integration point 5). The parser does not enforce it (a model over-emitting is tolerated and // merely over-counted); it bounds the prompt-side budget. const bankMaxLines = 12 // bankParserVersion versions the split/parse algorithm — the verdict-axis component folded into // banknoteSnap{enabled, parser_version} (§4в/§4а point 6), mirroring sanitizerSnap: a parser change // re-resolves the stripped draft, so it must be a loud --resnapshot even without a prompt edit. const bankParserVersion = "banknote-v1" // bankTypeOK is the accepted type set; anything else falls back to "term" (a benign default). var bankTypeOK = map[string]bool{"name": true, "place": true, "title": true, "term": true, "nickname": true} // bankFieldSplit tolerates a tab, a run of ≥2 spaces, or a pipe (optionally whitespace-padded) as the // field delimiter — the exact tolerance of banknote.py so a model that emits spaces instead of a real // TAB still parses. (Go \s is ASCII-whitespace; banknote lines are tab/space-delimited, so the parity // with Python's unicode \s around the pipe is exact on the term corpus.) var bankFieldSplit = regexp.MustCompile(`\t| {2,}|\s*\|\s*`) // bankEntry is one parsed candidate line (evidence for W1.5 §C; NOT written to the store until owner // signoff). Dst is the model's proposed translation — the direct dst delivery the co-occurrence // miner could not extract (蛊→гу, non-seed 龙公→Лун Гун). type bankEntry struct { Src string Dst string Type string } // bankFlags is the per-chunk telemetry surfaced loud at integration point 10. type bankFlags struct { NLines int // accepted (well-formed) banknote lines ParseFail bool // any malformed residual line (non-Han src / <2 fields) that was NOT a tolerated truncation Truncated bool // the LAST line was cut by generation length (tolerated, not a parse fail) } // splitBanknote slices the banknote block off the raw model output BEFORE any gate/editor (integration // point 1). Returns (clean_translation, raw_block). No separator → the whole (right-trimmed) output is // the clean translation and the block is "". Faithful to banknote.py: clean = output[:idx].rstrip(), // block = output[idx+len(SEP):].strip("\n"). func splitBanknote(output string) (clean, block string) { idx := strings.Index(output, bankSeparator) if idx < 0 { return strings.TrimRightFunc(output, unicode.IsSpace), "" } clean = strings.TrimRightFunc(output[:idx], unicode.IsSpace) block = strings.Trim(output[idx+len(bankSeparator):], "\n") return clean, block } // hasBankSrcHan reports whether src contains a CJK ideograph in the exact [㐀-鿿] range (U+3400–U+9FFF) // banknote.py checks — a zh→ru channel line whose src has no Han is malformed. Kept as the exact // Python range (NOT unicode.Han, which is wider) for byte-faithful parity. func hasBankSrcHan(src string) bool { for _, r := range src { if r >= 0x3400 && r <= 0x9FFF { return true } } return false } // parseBanknote parses the tab-delimited block (integration seam feeding §C evidence + telemetry). // Tolerant of a TRUNCATED final line when truncatedGeneration is set (§B3-5): a short last line under // truncation is flagged banknote_truncated, not counted as a parse fail. Any other malformed line // (fewer than 2 fields, or a src with no Han) sets banknote_parse_fail. Deterministic, no time/rand. func parseBanknote(block string, truncatedGeneration bool) ([]bankEntry, bankFlags) { var entries []bankEntry var flags bankFlags if strings.TrimSpace(block) == "" { return entries, flags } var lines []string for _, ln := range strings.Split(block, "\n") { if strings.TrimSpace(ln) != "" { lines = append(lines, ln) } } bad := 0 for i, ln := range lines { raw := bankFieldSplit.Split(strings.TrimSpace(ln), -1) var parts []string for _, p := range raw { if p = strings.TrimSpace(p); p != "" { parts = append(parts, p) } } if len(parts) < 2 { // A short LAST line under a truncated generation is a tolerated cut, not a failure. if i == len(lines)-1 && truncatedGeneration { flags.Truncated = true continue } bad++ continue } src, dst := parts[0], parts[1] typ := "term" if len(parts) >= 3 { typ = strings.ToLower(parts[2]) } if !bankTypeOK[typ] { typ = "term" } if !hasBankSrcHan(src) { bad++ continue } entries = append(entries, bankEntry{Src: src, Dst: dst, Type: typ}) } flags.NLines = len(entries) flags.ParseFail = bad > 0 return entries, flags }