textmachine/eval/bank_autonomy/m22b.py

63 lines
2.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""ЗАМОR 22b: classify (c) divergences semantic vs cosmetic; verify second-path mismatches. $0."""
from parse_common import parse_bankstop, parse_minedsig
import re, unicodedata
bs=parse_bankstop()
ms={r["src"]:r for r in parse_minedsig()}
n2={s:r for s,r in bs.items() if len({d for d,_ in r["drafts"]})>=2}
def norm(x):
x=x.lower().replace("«","").replace("»","").replace('"','').replace("'","")
x=re.sub(r"\s+"," ",x).strip()
x=x.replace("-"," ").replace("ё","е")
return re.sub(r"\s+"," ",x).strip()
# (c) refined
follow=cosmetic=semantic=0
sem_units=[]; cos_units=[]
for s,r in n2.items():
top=r["drafts"][0][0]; final=r["dst"]
if final==top: follow+=1; continue
if norm(final)==norm(top): cosmetic+=1; cos_units.append((s,r["type"],top,final))
else:
semantic+=1
# is final one of the OTHER drafts? (terminologist picked a lower-ranked variant)
in_drafts = any(norm(final)==norm(d) for d,_ in r["drafts"])
sem_units.append((s,r["type"],top,final,in_drafts))
print(f"=== (c) refined (N={len(n2)}) === follow(exact)={follow} cosmetic-only={cosmetic} SEMANTIC={semantic}")
# how many semantic finals ARE among the drafts vs invented-by-terminologist
in_d=sum(1 for *_,b in sem_units if b)
print(f" of {semantic} semantic overrides: final IS a lower-ranked draft: {in_d} ; final NOT in drafts (terminologist coined): {semantic-in_d}")
# 转-series systematic override check
zhuan=[(s,r["drafts"][0][0],r["dst"]) for s,r in n2.items() if s.endswith("") and len(s)<=3]
print("\n=== 转-series (rank) override ===")
for s,top,fin in sorted(zhuan):
print(f" {s}: §C2-3-top='{top}' -> FINAL='{fin}'")
print("\n=== SEMANTIC override units (final NOT in drafts = terminologist coined) ===")
for s,t,top,fin,ind in sem_units:
if not ind:
print(f" {s} [{t}]: top='{top}' -> FINAL='{fin}' (coined)")
# second-path: separate quote-escaping/truncation from genuine
print("\n=== SECOND-PATH mismatch diagnosis ===")
comparable=0; escaping=0; truncation=0; genuine=[]
for s,r in n2.items():
if s not in ms or not ms[s].get("other_proposals"): continue
comparable+=1
bs_others={norm(d) for d,_ in r["drafts"][1:]}
ms_props={norm(d) for d,_ in ms[s]["other_proposals"]}
if bs_others==ms_props: continue
# after normalization, is minedsig a subset of bankstop? => truncation
if ms_props<=bs_others:
truncation+=1
elif bs_others<=ms_props:
truncation+=1 # bankstop subset (other direction, still not genuine conflict of content)
else:
genuine.append((s,bs_others-ms_props,ms_props-bs_others))
print(f" comparable={comparable} truncation/subset(after norm)={truncation} GENUINE conflict={len(genuine)}")
for s,a,b in genuine:
print(f" {s}: only-in-bankstop={a} only-in-minedsig={b}")