113 lines
4.8 KiB
Python
113 lines
4.8 KiB
Python
#!/usr/bin/env python3
|
||
"""exp15 — S2' synthetic scene-marked slice via guzhenren chapter concatenation (owner decision 2026-07-17).
|
||
|
||
S2' = concatenate the FIRST N consecutive 节 of 蛊真人 (opening arc), STRIP the 第X节 headers at the seams,
|
||
so each seam is an IMPLICIT scene boundary (no explicit marker). Every seam is recorded as a ground-truth
|
||
boundary (rune-offset in the concatenated text + the two adjacent 节 ids). The seam previews (tail of 节_i +
|
||
head of 节_{i+1}) are emitted so each seam can be MANUALLY classified scene-shift vs continuation-cliffhanger
|
||
(§1.14b): continuation seams are negative-boundaries ("must NOT cut here"), separate strata for Q1.
|
||
|
||
Deterministic, $0, pure function of the book bytes. Outputs OUTSIDE git.
|
||
Run: eval/.venv/bin/python eval/exp15/build_s2prime.py [N] # default N=12 节
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import re
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
BOOK = Path("/home/ubuntu/books/gu-zhenren/guzhenren-utf8.txt")
|
||
OUT_DIR = Path("/home/ubuntu/books/gu-zhenren/exp15")
|
||
OUT_TXT = OUT_DIR / "S2prime.txt"
|
||
OUT_SEAMS = OUT_DIR / "S2prime_seams.json"
|
||
OUT_PREVIEW = OUT_DIR / "S2prime_seam_previews.md"
|
||
|
||
SEC_RE = re.compile(r"^第[0-9一二三四五六七八九十百千零两]{1,10}[节節][:: ]?.*$", re.M)
|
||
PREVIEW_TAIL, PREVIEW_HEAD = 220, 220
|
||
|
||
|
||
def load_text() -> str:
|
||
raw = BOOK.read_bytes()
|
||
try:
|
||
return raw.decode("utf-8")
|
||
except UnicodeDecodeError:
|
||
return raw.decode("gb18030", errors="replace")
|
||
|
||
|
||
def sections(text: str):
|
||
"""Return list of (header_line, body_text) for each 节 in file order (body = up to next 节)."""
|
||
marks = list(SEC_RE.finditer(text))
|
||
out = []
|
||
for k, m in enumerate(marks):
|
||
header = m.group(0).strip()
|
||
body_start = m.end()
|
||
body_end = marks[k + 1].start() if k + 1 < len(marks) else len(text)
|
||
body = text[body_start:body_end].strip("\n")
|
||
out.append((k + 1, header, body)) # sec index n = k+1 (running, per gu_corpus_build convention)
|
||
return out
|
||
|
||
|
||
def main():
|
||
n = int(sys.argv[1]) if len(sys.argv) > 1 else 12
|
||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
text = load_text()
|
||
secs = sections(text)
|
||
if len(secs) < n:
|
||
print(f"only {len(secs)} 节 found; using all", file=sys.stderr)
|
||
n = len(secs)
|
||
chosen = secs[:n]
|
||
|
||
# concatenate bodies (headers STRIPPED at seams); single "\n" join = the implicit seam
|
||
parts, seams, cursor = [], [], 0
|
||
JOIN = "\n"
|
||
for i, (idx, header, body) in enumerate(chosen):
|
||
if i > 0:
|
||
cursor += len(JOIN) # account the join separator (rune count)
|
||
seam_offset = cursor # rune offset where this section's body begins
|
||
if i > 0:
|
||
seams.append({
|
||
"seam_id": i,
|
||
"rune_offset": seam_offset, # boundary between prev body and this body
|
||
"prev_sec": chosen[i - 1][0],
|
||
"next_sec": idx,
|
||
"prev_header": chosen[i - 1][1],
|
||
"next_header": header,
|
||
"class": "UNCLASSIFIED", # -> scene-shift | continuation (manual, §1.14b)
|
||
"rationale": "",
|
||
})
|
||
parts.append(body)
|
||
cursor += len(body)
|
||
joined = JOIN.join(parts)
|
||
|
||
OUT_TXT.write_text(joined, encoding="utf-8")
|
||
OUT_SEAMS.write_text(json.dumps({
|
||
"book": "guzhenren", "arc": "opening", "n_sections": n,
|
||
"section_ids": [c[0] for c in chosen],
|
||
"total_runes": len(joined),
|
||
"headers_stripped_at_seams": True,
|
||
"seam_marker": "implicit (single \\n; no 第X节 header)",
|
||
"seams": seams,
|
||
}, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
|
||
# seam-context previews for MANUAL classification (read tail_i + head_{i+1})
|
||
pv = [f"# S2' seam previews — classify each: scene-shift | continuation-cliffhanger (§1.14b)",
|
||
f"# {n} 节 concatenated, headers stripped; {len(seams)} seams; total {len(joined):,} runes", ""]
|
||
for s in seams:
|
||
off = s["rune_offset"]
|
||
tail = joined[max(0, off - len(JOIN) - PREVIEW_TAIL):off - len(JOIN)]
|
||
head = joined[off:off + PREVIEW_HEAD]
|
||
pv += [f"## seam {s['seam_id']} (节{s['prev_sec']} → 节{s['next_sec']}, offset {off})",
|
||
f"**{s['prev_header']}** → **{s['next_header']}**",
|
||
f"…tail of 节{s['prev_sec']}: …{tail}",
|
||
f"head of 节{s['next_sec']}: {head}…", ""]
|
||
OUT_PREVIEW.write_text("\n".join(pv), encoding="utf-8")
|
||
|
||
print(f"S2' built: {n} 节, {len(seams)} seams, {len(joined):,} runes -> {OUT_TXT}")
|
||
print(f"section ids: {[c[0] for c in chosen]}")
|
||
print(f"seams (offset, 节i->节j): " + ", ".join(f"{s['rune_offset']}({s['prev_sec']}->{s['next_sec']})" for s in seams))
|
||
print(f"previews for manual classification -> {OUT_PREVIEW}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|