textmachine/backend/internal/seed/normalize.go

128 lines
5.1 KiB
Go

package seed
import "strings"
// normalize.go: the CANONICAL FORM of a document of this schema — the shape the engine WRITES.
//
// Two jobs, and the second is the reason this file exists rather than a `strings.TrimSpace` at a call
// site.
//
// (1) ONE RULE FOR ALL TEXT. The bank loader already trims the fields it keys on (`src`, `sense`,
// `dst` — memseed.go says why: an untrimmed key never matches and the term is silently inert), and the
// decisions door trims the same three. Every other text field escaped that rule only because nothing
// keyed on it. Stated once over the schema, a field added later inherits it instead of becoming the
// next exception.
//
// (2) IT IS WHAT KEEPS THE DOCUMENT WRITABLE. Measured on gopkg.in/yaml.v3 v3.0.1 (the version in
// go.mod): a string whose first byte is a newline is emitted as a block scalar carrying an explicit
// indentation indicator that does not match the indentation the emitter actually wrote inside a
// sequence item, and the library's OWN parser then either refuses the document —
//
// note: |4-
// не термин → yaml: line 1: did not find expected key
//
// — or reads a DIFFERENT value back: `"\n"` returns as `""`, `"\n x"` as `"x"`. Both documents of this
// schema are sequences of mappings, so this reaches EVERY text field of both, not one field of one.
// Trimming removes the leading newline; it does not remove the class, which is why membank's render
// PROVES its output instead of trusting this (see membank.RenderSeedFile).
//
// Normalizing is deliberately NOT done on the read path. The decoders answer "what does this file
// say"; changing that would change what a run reads out of every book's glossary_seed, which is a
// different decision from what the decisions door writes back.
// normText is the one rule. Surrounding whitespace in a data field is not data: it never survives the
// bank loader's keying, and leading whitespace that starts with a newline is what the emitter cannot
// write back.
func normText(s string) string { return strings.TrimSpace(s) }
// normList trims every member and returns nil for an empty list, so a document that came off disk and
// the same document rendered and read back compare equal (a YAML `[]` decodes to an empty slice, an
// absent key to a nil one, and the format has no third state).
func normList(in []string) []string {
if len(in) == 0 {
return nil
}
out := make([]string, len(in))
for i, s := range in {
out[i] = normText(s)
}
return out
}
// Normalize returns the document in canonical form. It is a COPY: the caller's slices are shared with
// whatever loaded them, and a normalizer that wrote through them would edit a document somebody else is
// still comparing against.
func (f File) Normalize() File {
out := File{}
if len(f.Terms) > 0 {
out.Terms = make([]Term, len(f.Terms))
for i, t := range f.Terms {
out.Terms[i] = t.Normalize()
}
}
if len(f.Voices) > 0 {
out.Voices = make([]Voice, len(f.Voices))
for i, v := range f.Voices {
out.Voices[i] = v.Normalize()
}
}
if len(f.Addresses) > 0 {
out.Addresses = make([]Address, len(f.Addresses))
for i, a := range f.Addresses {
out.Addresses[i] = a.Normalize()
}
}
return out
}
// Normalize returns the term in canonical form.
func (t Term) Normalize() Term {
t.Src, t.Dst, t.Type, t.Sense = normText(t.Src), normText(t.Dst), normText(t.Type), normText(t.Sense)
t.Gender, t.Speech = normText(t.Gender), normText(t.Speech)
t.TranslitPolicy, t.FirstPerson = normText(t.TranslitPolicy), normText(t.FirstPerson)
t.NicknameTranslation, t.Status, t.Note = normText(t.NicknameTranslation), normText(t.Status), normText(t.Note)
if t.Decl != nil {
d := Decl{Invariant: t.Decl.Invariant, Forms: normList(t.Decl.Forms)}
t.Decl = &d
}
if len(t.Aliases) == 0 {
t.Aliases = nil
return t
}
aliases := make([]Alias, len(t.Aliases))
for i, a := range t.Aliases {
aliases[i] = Alias{Alias: normText(a.Alias), Type: normText(a.Type)}
}
t.Aliases = aliases
return t
}
// Normalize returns the voice profile in canonical form.
func (v Voice) Normalize() Voice {
v.Src, v.Sense, v.Register = normText(v.Src), normText(v.Sense), normText(v.Register)
v.SelfRef, v.AddressDefault = normText(v.SelfRef), normText(v.AddressDefault)
v.Brightness = normText(v.Brightness)
v.LexiconMarkers, v.NGLexicon, v.Exemplars = normList(v.LexiconMarkers), normList(v.NGLexicon), normList(v.Exemplars)
return v
}
// Normalize returns the address-pair record in canonical form.
func (a Address) Normalize() Address {
a.Speaker, a.SpeakerSense = normText(a.Speaker), normText(a.SpeakerSense)
a.Addressee, a.AddresseeSense = normText(a.Addressee), normText(a.AddresseeSense)
a.Register, a.Form, a.Closeness = normText(a.Register), normText(a.Form), normText(a.Closeness)
return a
}
// Normalize returns the reject list in canonical form.
func (f RejectFile) Normalize() RejectFile {
out := RejectFile{}
if len(f.Rejects) == 0 {
return out
}
out.Rejects = make([]Reject, len(f.Rejects))
for i, r := range f.Rejects {
out.Rejects[i] = Reject{Src: normText(r.Src), Note: normText(r.Note)}
}
return out
}