#!/usr/bin/env python3 """金瓶梅 (Jin Ping Mei, Ming-dynasty PUBLIC DOMAIN) l2-erotica corpus builder for the D14.4 erotica arm (exp11 / exp10 Erotica section, zh→ru). Sibling of gu_corpus_build.py (violence). The book text and derivatives are OUTSIDE git (data/ is gitignored); only the SELECTED fragments become corpus rows. Source: ctext.org simplified (/jinpingmei/chNN/zhs), fetched into a working .txt with '===== 第N回 =====' markers (graphics = SIMPLIFIED, verified — trad2simp NOT needed). This helper: - splits the working text into 回 (chapters) by the marker; - scores each 回 / paragraph by a CLASSICAL erotic-marker density (metadata only — never dumps bodies); - extracts a paragraph-bounded fragment (~target chars) from a chosen 回 at a chosen offset; - writes JSONL records in the refusal_corpus schema (id-prefix l2-ero-zh-jpm). ADULT / LEVEL-3 SCREEN (mandate, per-participant): Jin Ping Mei's sexual scenes are between ADULTS (西门庆 with 潘金莲 / 李瓶儿 / 王六儿 / 宋惠莲 / 如意儿 — all adult married women). The maid 春梅 (Chunmei) is age-ambiguous in early chapters, so 春梅-centric early explicit scenes are DELIBERATELY excluded. Every selected fragment is screened by named participants against the known adult cast; a fragment with any minor / age-ambiguous participant in a sexual context is dropped (level 3 — never processed). Usage: eval/.venv/bin/python eval/jpm_corpus_build.py index eval/.venv/bin/python eval/jpm_corpus_build.py candidates # per-回 density (metadata) eval/.venv/bin/python eval/jpm_corpus_build.py paras --hui 27 # per-paragraph density in a 回 eval/.venv/bin/python eval/jpm_corpus_build.py show --hui 27 --start 8 --chars 1500 # ONE fragment (review) eval/.venv/bin/python eval/jpm_corpus_build.py extract --spec spec.json --out .jsonl """ from __future__ import annotations import argparse, json, re, sys from pathlib import Path BOOK = Path("/home/ubuntu/books/jinpingmei/jinpingmei-ctext-zhs.txt") HUI_RE = re.compile(r"^=====\s*第(\d+)回\s*=====") # Classical erotic-marker lexicon (Simplified). Weighted; density = weighted hits / chars. # Deliberately excludes name-colliding tokens (金莲 = 潘金莲's name) and weak generic tokens. ERO_LEX = { # acts (classical / euphemistic) "云雨": 4, "交欢": 5, "交媾": 5, "交接": 3, "采战": 5, "抽送": 5, "品箫": 5, "耸": 2, "云情雨意": 5, "颠鸾倒凤": 5, "鏖战": 3, "偷情": 3, "苟合": 4, # anatomy (classical) "麈柄": 5, "龟头": 4, "阳物": 4, "玉茎": 4, "牝": 3, "阴户": 5, "阴": 1, "花心": 3, "红绫": 1, "托子": 3, "银托": 3, "勉铃": 4, # fluids / heat "淫水": 5, "淫津": 5, "淫": 2, "精": 1, "欲火": 3, "情欲": 3, "春意": 3, "春心": 2, # imagery / bodies "云鬟": 1, "香肌": 2, "玉体": 3, "酥胸": 3, "罗袜": 1, "云雨才罢": 5, "巫山": 3, "鸾凤": 2, "欢娱": 2, "亵": 2, "媱": 3, "淫妇": 3, } def load_hui() -> list[dict]: if not BOOK.exists(): sys.exit(f"missing working text: {BOOK}\n run the ctext fetcher first (scratchpad/jpm_fetch.py).") lines = BOOK.read_text(encoding="utf-8").split("\n") marks = [(i, m.group(1)) for i, l in enumerate(lines) if (m := HUI_RE.match(l.strip()))] hui = [] for k, (i, num) in enumerate(marks): end = marks[k + 1][0] if k + 1 < len(marks) else len(lines) body = "\n".join(x for x in lines[i + 1:end] if x.strip()).strip() hui.append({"hui": int(num), "line": i, "body": body, "chars": len(body)}) return hui def score(body: str) -> tuple[float, dict]: hits, total = {}, 0 for kw, w in ERO_LEX.items(): c = body.count(kw) if c: hits[kw] = c total += c * w return total / max(1, len(body)), hits def paras(body: str) -> list[str]: return [p.strip() for p in re.split(r"\n", body) if p.strip()] def frag_from(body: str, start_para: int, target: int) -> str: ps = paras(body) buf, tot = [], 0 for p in ps[start_para:]: if buf and tot + len(p) > target: break buf.append(p); tot += len(p) return "\n".join(buf) if buf else body[:target] def cmd_index(_): h = load_hui() print(f"回 loaded: {len(h)} ({', '.join(str(x['hui']) for x in h)})") print(f"total chars: {sum(x['chars'] for x in h):,}") def cmd_candidates(args): h = load_hui() scored = [] for s in h: d, hits = score(s["body"]) top = sorted(hits.items(), key=lambda kv: -kv[1] * ERO_LEX[kv[0]])[:6] scored.append((d, s["hui"], s["chars"], top)) scored.sort(key=lambda x: -x[0]) print("# density | 回 | chars | top-markers") for d, hui, chars, top in scored: kw = " ".join(f"{k}×{c}" for k, c in top) print(f"{d:.4f} | {hui:3} | {chars:6} | {kw}") def cmd_paras(args): h = {x["hui"]: x for x in load_hui()}[args.hui] ps = paras(h["body"]) print(f"# 回{args.hui}: {len(ps)} paragraphs") print("# idx | chars | density | top-markers | preview(head 12 han)") for i, p in enumerate(ps): d, hits = score(p) top = sorted(hits.items(), key=lambda kv: -kv[1] * ERO_LEX[kv[0]])[:4] kw = " ".join(f"{k}×{c}" for k, c in top) head = "".join(re.findall(r"[一-鿿]", p))[:12] print(f"{i:3} | {len(p):5} | {d:.4f} | {kw:24} | {head}") def cmd_show(args): h = {x["hui"]: x for x in load_hui()}[args.hui] frag = frag_from(h["body"], args.start, args.chars) d, hits = score(frag) print(f"# 回{args.hui} start-para {args.start} ({len(frag)} chars) density={d:.4f}") print(f"# markers: {sorted(hits.items(), key=lambda kv:-kv[1]*ERO_LEX[kv[0]])}") print(frag) def cmd_extract(args): """spec.json: [{"id","hui","start","chars","note","participants"}...] → JSONL refusal_corpus rows. 'participants' documents the adult-screen (named adult cast) inline in the note.""" hui = {x["hui"]: x for x in load_hui()} spec = json.loads(Path(args.spec).read_text(encoding="utf-8")) out = [] for item in spec: s = hui[item["hui"]] frag = frag_from(s["body"], item.get("start", 0), item.get("chars", 1500)) d, _ = score(frag) note = item.get("note", "") if item.get("participants"): note = (note + f" | ADULT-SCREEN participants: {item['participants']}").strip(" |") out.append({"id": item["id"], "lang": "zh", "level": 2, "category": "erotica", "source": f"金瓶梅 (Ming, PD) 第{item['hui']}回, ctext.org simplified, вне git", "license": "public domain (Ming dynasty); ctext.org/jinpingmei/zhs", "expected": "translate", "density": round(d, 4), "note": note, "text": frag}) Path(args.out).write_text("\n".join(json.dumps(r, ensure_ascii=False) for r in out) + "\n", encoding="utf-8") print(f"wrote {len(out)} records → {args.out}") for r in out: print(f" {r['id']} 回-based {len(r['text'])} chars density={r['density']} [{r['category']}]") def main(): ap = argparse.ArgumentParser() sub = ap.add_subparsers(dest="cmd", required=True) sub.add_parser("index").set_defaults(fn=cmd_index) sub.add_parser("candidates").set_defaults(fn=cmd_candidates) pp = sub.add_parser("paras"); pp.add_argument("--hui", type=int, required=True); pp.set_defaults(fn=cmd_paras) sh = sub.add_parser("show"); sh.add_argument("--hui", type=int, required=True) sh.add_argument("--start", type=int, default=0); sh.add_argument("--chars", type=int, default=1500) sh.set_defaults(fn=cmd_show) ex = sub.add_parser("extract"); ex.add_argument("--spec", required=True); ex.add_argument("--out", required=True) ex.set_defaults(fn=cmd_extract) args = ap.parse_args() args.fn(args) if __name__ == "__main__": main()