textmachine/eval/design11/ws2_chunk_sim.py

176 lines
7.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""WS2 (г) verification — chunking + edit-unit simulation on the 25-chapter book.
Reconstructs per-chapter source from rerun/records.json, then simulates:
(1) DRAFT chunking under the OUTPUT-token budget (fertility est_out), for target_out
in {1500, 1797, 2200}, vs the CURRENT input-est greedy budget (targetChunkTokens=1500);
(2) EDIT-UNIT = chapter (with paragraph-split when a chapter exceeds a ceiling).
Reports: est_out histograms, share in the 2k-8k-out degradation zone (research/19 §A.1.1),
floor-invariant (chunks under MinChunkChars=500 src chars), chapters>ceiling, chunk/unit counts,
and a COGS forecast via fertility (flash draft + glm-5 edit).
$0: deterministic, no LLM, no network. Reuses eval/exp15/chunker.py greedy+sentence helpers.
Writes /home/ubuntu/books/gu-zhenren/design11/ws2_chunk_sim.json (OUT of git).
"""
import json, os, sys
from collections import defaultdict, Counter
sys.path.insert(0, "/home/ubuntu/projects/textmachine/eval/exp15")
import chunker as CK # greedy_chunks, est_out, _est_tokens_from, _class_counts, _pack_sentences, split_paragraphs, normalize_source
RECORDS = "/home/ubuntu/books/gu-zhenren/rerun/records.json"
OUT = "/home/ubuntu/books/gu-zhenren/design11/ws2_chunk_sim.json"
F_CJK, F_OTHER = 1.1978, 0.3852
MIN_CHUNK_CHARS = 500 # Gates.Coverage.MinChunkChars (pipeline.go:159)
DEG_LO, DEG_HI = 2000, 8000 # emission degradation zone, output tokens (research/19 §A.1.1)
# prices USD/1M (freeze_facts / vendor-verified): draft flash out, edit glm-5 out (thinking-off)
PRICE = {"flash_in": 0.14, "flash_cache": 0.0028, "flash_out": 0.28,
"glm5_in": 1.00, "glm5_cache": 0.20, "glm5_out": 3.20}
def est_out(text):
return CK.est_out(text) # F_CJK*cjk + F_OTHER*other on source chars
def build_chapters():
recs = json.load(open(RECORDS))
by_ch = defaultdict(list)
for r in recs:
by_ch[r["chapter"]].append((r["chunk_idx"], r.get("source", "") or ""))
chapters = {}
for ch, items in by_ch.items():
items.sort()
chapters[ch] = "\n".join(s for _, s in items)
return dict(sorted(chapters.items()))
def greedy_out(text, target_out):
"""Greedy paragraph pack to an OUTPUT-token budget (fertility est_out), mirroring
appendChapterChunks but with the budget in ru-output tokens. Oversize paragraph ->
sentence pack; oversize sentence -> own chunk (never split a sentence)."""
chapter = CK.normalize_source(text)
paras = CK.split_paragraphs(chapter)
chunks, buf = [], []
def buf_out():
return est_out("\n\n".join(buf)) if buf else 0.0
def flush():
nonlocal buf
if buf:
t = "\n\n".join(buf).strip()
if t:
chunks.append(t)
buf = []
for p in paras:
if est_out(p) > target_out:
flush()
# sentence pack this oversize paragraph to target_out
segs = CK.split_source_sentences(p)
sbuf = []
def sflush():
nonlocal sbuf
if sbuf:
chunks.append("".join(sbuf))
sbuf = []
for s in segs:
if sbuf and est_out("".join(sbuf) + s) > target_out:
sflush()
sbuf.append(s)
sflush()
continue
if buf and est_out("\n\n".join(buf) + "\n\n" + p) > target_out:
flush()
buf.append(p)
flush()
return chunks
def nonspace_chars(s):
return sum(1 for c in s if not c.isspace())
def hist(values, edges):
c = Counter()
for v in values:
placed = False
for lo, hi in edges:
if lo <= v < hi:
c[f"{lo}-{hi}"] += 1; placed = True; break
if not placed:
c[f">={edges[-1][1]}"] += 1
return dict(c)
def summarize_chunks(chunks):
outs = [est_out(c) for c in chunks]
src_chars = [nonspace_chars(c) for c in chunks]
return {
"n_chunks": len(chunks),
"est_out_mean": round(sum(outs)/len(outs), 1) if outs else 0,
"est_out_min": round(min(outs), 1) if outs else 0,
"est_out_max": round(max(outs), 1) if outs else 0,
"under_min_chunk_chars(500)": sum(1 for s in src_chars if s < MIN_CHUNK_CHARS),
"in_degradation_zone_2k_8k": sum(1 for o in outs if DEG_LO <= o <= DEG_HI),
"above_deg_hi(8k)": sum(1 for o in outs if o > DEG_HI),
"hist_est_out": hist(outs, [(0,1000),(1000,2000),(2000,3200),(3200,5000),(5000,8000)]),
"total_est_out": round(sum(outs)),
}
def main():
chapters = build_chapters()
result = {"n_chapters": len(chapters), "MIN_CHUNK_CHARS": MIN_CHUNK_CHARS,
"degradation_zone_out_tokens": [DEG_LO, DEG_HI]}
# --- (a) CURRENT chunking: greedy on input-est targetChunkTokens=1500 (chunker.go behavior) ---
cur_chunks = []
for ch, txt in chapters.items():
cur_chunks += CK.greedy_chunks(txt) # input-est 1500 budget (backend const)
result["current_greedy_input_est_1500"] = summarize_chunks(cur_chunks)
# --- (b) DRAFT chunking under OUTPUT-token budget (fertility), swept target_out ---
result["draft_output_budget"] = {}
for tgt in (1500, 1797, 2200):
dchunks = []
for ch, txt in chapters.items():
dchunks += greedy_out(txt, tgt)
result["draft_output_budget"][f"target_out_{tgt}"] = summarize_chunks(dchunks)
# --- (c) EDIT-UNIT = chapter; paragraph-split when chapter est_out > ceiling ---
for ceiling in (8000, 3200): # 8000 = gated large-chapter arm; 3200 = ratified default (review-2 F1)
units = []
chapters_over = 0
split_units = 0
for ch, txt in chapters.items():
co = est_out(txt)
if co <= ceiling:
units.append(txt)
else:
chapters_over += 1
# split at paragraph boundaries greedily to <= ceiling (never split a paragraph)
sub = greedy_out(txt, ceiling)
split_units += len(sub)
units += sub
s = summarize_chunks(units)
s["chapters_over_ceiling"] = chapters_over
s["units_from_split_chapters"] = split_units
s["chapter_est_out_raw"] = {str(ch): round(est_out(txt)) for ch, txt in chapters.items()}
result[f"edit_unit_chapter_ceiling_{ceiling}"] = s
# --- (d) COGS forecast per structure (input via records source, output via fertility) ---
# draft: input = source est-tokens; output = est_out. edit: input = source+draft (~2x src) est; output ~= 0.85*draft_out (edit shrinks slightly)
total_src_est_in = sum(CK._est_tokens_from(*CK._class_counts(txt)) for txt in chapters.values())
total_draft_out = sum(est_out(txt) for txt in chapters.values())
# naive per-book COGS on the whole book (no cache) as an upper bound; cache handled in WS1/WS6
draft_in_cost = total_src_est_in * PRICE["flash_in"] / 1e6
draft_out_cost = total_draft_out * PRICE["flash_out"] / 1e6
edit_in_est = (total_src_est_in + total_draft_out) # editor sees src + draft
edit_in_cost = edit_in_est * PRICE["glm5_in"] / 1e6
edit_out_cost = (total_draft_out * 0.9) * PRICE["glm5_out"] / 1e6 # edit ~= draft length
result["cogs_forecast_25ch_no_cache_upper_bound"] = {
"total_src_est_in_tokens": round(total_src_est_in),
"total_draft_out_tokens": round(total_draft_out),
"draft_cost_usd": round(draft_in_cost + draft_out_cost, 4),
"edit_cost_usd": round(edit_in_cost + edit_out_cost, 4),
"total_usd": round(draft_in_cost + draft_out_cost + edit_in_cost + edit_out_cost, 4),
"note": "upper bound, no prefix cache; WS1/WS6 apply warm-then-fan cache (-80% cached input)",
}
os.makedirs(os.path.dirname(OUT), exist_ok=True)
json.dump(result, open(OUT, "w"), ensure_ascii=False, indent=2)
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()