178 lines
8.3 KiB
Go
178 lines
8.3 KiB
Go
package pipeline
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"testing"
|
|
|
|
"textmachine/backend/internal/config"
|
|
)
|
|
|
|
// promptlabel_test.go: `prompt_version` must track the BYTES of the prompt it labels.
|
|
//
|
|
// WHY THIS IS A GATE AND NOT A NORM. The norm already exists, written into the shipping config itself —
|
|
// pipeline-c2.yaml: «лейбл обязан следовать за новым SHA файла» — and it was broken anyway:
|
|
// prompts/zh-ru/editor.md was edited on 2026-08-01 while prompt_version stayed v3-discourse-reflow. The
|
|
// cold run of 31.08 then nearly built a causal conclusion on comparing itself against a July bench
|
|
// carrying that same label. Money was never at risk (PromptSHA256 folds into the snapshot, so the ENGINE
|
|
// always knew); what was at risk is the COMPARABILITY of two runs, which is what an experiment is made of.
|
|
// A rule that lives only in a comment inside a ratified decision, and is then skipped by the next ratified
|
|
// decision, is a rule that needs a machine.
|
|
//
|
|
// WHY A REPO GATE AND NOT A RUNTIME CHECK. A runtime check can only compare a label against the payloads
|
|
// THIS project database happens to hold, so it catches a label reused within one book. The incident was
|
|
// BETWEEN books — a bench project and a fresh one, different databases — which is exactly the case a
|
|
// runtime check cannot see. The repository is where both live, so the repository is where the pair is
|
|
// pinned. It is also $0, snapshot-neutral, and it fails at CI time rather than after a purchase.
|
|
//
|
|
// UPDATING IT IS THE POINT, NOT AN OBSTACLE: a deliberate prompt edit is expected to come with a new
|
|
// label, and `TM_UPDATE_PROMPT_LABELS=1 go test ./internal/pipeline/ -run TestPromptLabelsPinTheirBytes`
|
|
// re-writes the ledger — the same shape as TM_UPDATE_GOLDEN. What must never happen silently is the third
|
|
// case: the same label over different bytes.
|
|
|
|
// promptLabelLedger is the checked-in memory: "<pair>/<role>/<label>" → sha256 of the CANONICAL prompt
|
|
// (comments stripped — the same form the snapshot folds, so an edited comment costs nobody a re-purchase).
|
|
const promptLabelLedger = "testdata/prompt-labels.json"
|
|
|
|
// shippingPipelines are the configs whose prompts ship. The arms are included because an arm exists to
|
|
// isolate ONE variable, and a prompt that moved under a stale label puts a second one in the comparison.
|
|
var shippingPipelines = []string{
|
|
"pipeline-c1.yaml", "pipeline-c2.yaml",
|
|
"pipeline-arm-deepseek-pro.yaml", "pipeline-arm-glm.yaml", "pipeline-arm-mistral.yaml",
|
|
}
|
|
|
|
func TestPromptLabelsPinTheirBytes(t *testing.T) {
|
|
models, err := config.LoadModels(filepath.Join("..", "..", "configs", "models.yaml"))
|
|
if err != nil {
|
|
t.Fatalf("load the shipping models.yaml: %v", err)
|
|
}
|
|
const pair = "zh-ru" // the only pair with a prompt pack in the repo; a new pair joins this map by existing
|
|
seen := map[string]string{}
|
|
for _, pf := range shippingPipelines {
|
|
p, err := config.LoadPipeline(filepath.Join("..", "..", "configs", pf), models, pair, nil)
|
|
if err != nil {
|
|
t.Fatalf("load %s: %v", pf, err)
|
|
}
|
|
for _, st := range p.Stages {
|
|
key := fmt.Sprintf("%s/%s/%s", pair, st.Role, st.PromptVersion)
|
|
tpl, err := LoadPromptTemplate(st.PromptPath)
|
|
if err != nil {
|
|
t.Fatalf("%s stage %q: load its resolved prompt %s: %v", pf, st.Name, st.PromptPath, err)
|
|
}
|
|
// A label used by two stages must mean ONE file's bytes, or it is not a label.
|
|
if prev, dup := seen[key]; dup && prev != tpl.SHA256 {
|
|
t.Errorf("label %q names two different prompt bodies across the shipping configs (%s vs %s) — "+
|
|
"a label that is not a function of the bytes cannot make two runs comparable",
|
|
key, prev[:12], tpl.SHA256[:12])
|
|
}
|
|
seen[key] = tpl.SHA256
|
|
}
|
|
// The BANK ROLES resolve their prompts through the gate rather than through a stage, and they are
|
|
// paid calls whose comparability matters exactly as much: the terminologist's consolidation is what
|
|
// a book's bank ends up saying. They carry no prompt_version of their own, so the gate's own
|
|
// version string is the label — which is the honest answer, not a workaround: it is what a run logs
|
|
// as the identity of that contour.
|
|
//
|
|
// ⚠ BOTH bank roles, not one. The classifier was outside this sweep while no shipping config turned
|
|
// it on; backlog row 140 turns it on, and its prompt decides a term's CLASS — which routes a name to
|
|
// transliteration and a realia to a translation. A pass whose class definitions moved under a
|
|
// stationary label is exactly as uncomparable as a render whose instructions did.
|
|
if g := p.Gates.Terminology; g.Enabled {
|
|
for _, role := range []struct{ name, path string }{
|
|
{"terminologist", g.PromptPath},
|
|
{"classifier", g.ClassifyPromptPath},
|
|
} {
|
|
if role.path == "" {
|
|
continue // the phase is off, so it has no prompt and buys nothing
|
|
}
|
|
tpl, err := LoadPromptTemplate(role.path)
|
|
if err != nil {
|
|
t.Fatalf("%s: load the %s prompt %s: %v", pf, role.name, role.path, err)
|
|
}
|
|
seen[fmt.Sprintf("%s/%s/%s", pair, role.name, terminologyVersion)] = tpl.SHA256
|
|
}
|
|
}
|
|
}
|
|
if len(seen) == 0 {
|
|
t.Fatal("no shipping stage resolved a prompt — this gate would be enforcing nothing")
|
|
}
|
|
|
|
if os.Getenv("TM_UPDATE_PROMPT_LABELS") == "1" {
|
|
writePromptLedger(t, seen)
|
|
t.Logf("prompt-label ledger re-written with %d entr(ies) — review the diff: a NEW key is a "+
|
|
"deliberate bump, a CHANGED value under an existing key is the defect this gate exists for", len(seen))
|
|
return
|
|
}
|
|
|
|
raw, err := os.ReadFile(promptLabelLedger)
|
|
if err != nil {
|
|
t.Fatalf("read the prompt-label ledger (%s): %v — create it with TM_UPDATE_PROMPT_LABELS=1", promptLabelLedger, err)
|
|
}
|
|
var stored map[string]string
|
|
if err := json.Unmarshal(raw, &stored); err != nil {
|
|
t.Fatalf("the prompt-label ledger is not readable JSON: %v", err)
|
|
}
|
|
// ⚠ AN ENTRY NOBODY CLAIMED IS A LOSS, NOT A LEFTOVER, and the sweep could not see one: the loop below
|
|
// walks `seen` and never `stored`, so DROPPING a role from the sweep above left its ledger entry orphaned
|
|
// and its prompt unpinned, with every test green. Found by planting — the classifier line was added to
|
|
// this sweep by the pack that made it ship, and removing it again cost nothing.
|
|
//
|
|
// A legitimately retired label is retired by deleting its line here in the same commit; an orphan means
|
|
// the sweep stopped looking at something it used to watch.
|
|
for _, key := range sortedKeys(stored) {
|
|
if _, claimed := seen[key]; !claimed {
|
|
t.Errorf("the ledger records %q and this run found nothing that uses it — either a shipping config stopped resolving that prompt, or the sweep above stopped covering it. Delete the line deliberately, or restore what dropped out.", key)
|
|
}
|
|
}
|
|
for _, key := range sortedKeys(seen) {
|
|
want, known := stored[key]
|
|
switch {
|
|
case !known:
|
|
t.Errorf("label %q is not in the ledger. If you BUMPED the label deliberately, re-write the "+
|
|
"ledger with TM_UPDATE_PROMPT_LABELS=1 and commit it — that is the whole point. If you did "+
|
|
"not, a stage is quoting a label nobody recorded.", key)
|
|
case want != seen[key]:
|
|
t.Errorf("⚠ THE PROMPT MOVED AND ITS LABEL DID NOT. %q was recorded over sha %s and now resolves "+
|
|
"to %s. Two runs carrying this label are NOT comparable, and nothing else in the repository "+
|
|
"would have said so (prompts/zh-ru/editor.md, 2026-08-01, is this defect's own history). "+
|
|
"Bump prompt_version in the shipping config and re-write the ledger with "+
|
|
"TM_UPDATE_PROMPT_LABELS=1 — or restore the bytes.", key, want[:12], seen[key][:12])
|
|
}
|
|
}
|
|
}
|
|
|
|
func sortedKeys(m map[string]string) []string {
|
|
out := make([]string, 0, len(m))
|
|
for k := range m {
|
|
out = append(out, k)
|
|
}
|
|
sort.Strings(out)
|
|
return out
|
|
}
|
|
|
|
// writePromptLedger renders the ledger deterministically: sorted keys, indented, trailing newline — so a
|
|
// re-write produces a diff a human reads rather than a re-ordering.
|
|
func writePromptLedger(t *testing.T, seen map[string]string) {
|
|
t.Helper()
|
|
var b strings.Builder
|
|
b.WriteString("{\n")
|
|
keys := sortedKeys(seen)
|
|
for i, k := range keys {
|
|
comma := ","
|
|
if i == len(keys)-1 {
|
|
comma = ""
|
|
}
|
|
fmt.Fprintf(&b, " %q: %q%s\n", k, seen[k], comma)
|
|
}
|
|
b.WriteString("}\n")
|
|
if err := os.MkdirAll(filepath.Dir(promptLabelLedger), 0o755); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := os.WriteFile(promptLabelLedger, []byte(b.String()), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}
|