package pgstore import ( "context" "errors" "fmt" "time" "github.com/jackc/pgx/v5" "github.com/jackc/pgx/v5/pgconn" "textmachine/platform/internal/money" ) // Tx is a transaction of this store. Exported so a caller can join its own write to one — the queue // does, because a run row and its queue entry must be written together — without every package in // the zone importing pgx to name the type. type Tx = pgx.Tx var ( // ErrRunInFlight is a second live run on one book. The database refuses it (the partial unique // index runs_one_live_per_book) rather than the API explaining it: the engine holds an EXCLUSIVE // lock on the project file, so two live runs is not a state anything downstream can represent. ErrRunInFlight = errors.New("pgstore: the book already has a live run") // ErrNoRun is a run that does not exist or does not belong to the caller. ErrNoRun = errors.New("pgstore: no such run") ) // StartRunInput is one accepted run request. type StartRunInput struct { UserID string BookID string VerifyBank bool CeilingChapters int // Ceiling is what those chapters are worth and what is HELD before anything is spawned. The hold // and the ceiling handed to the engine are the same number by construction: the hold makes the // credit unavailable to any other run, and the engine stops itself there, so an overspend is // impossible even while the platform is blind (D39.100). Ceiling money.MicroUSD Now time.Time } // StartedRun is what the caller needs after a successful start. type StartedRun struct { Run AttemptID int64 // JournalOffset is the size of the book's journal at the moment this attempt was admitted. The // journal is per BOOK and append-only, so an attempt's own lines begin after everything already // in it; without this the tailer would adopt the previous attempt's handshake. JournalOffset int64 } // StartRun admits a run: the run row, its first attempt, the hold and the queue entry, all in ONE // transaction. // // One transaction because each of them alone is a way to lose money or work. A hold without a run // row is credit reserved for nothing; a run row without a hold is a run that spends credit nobody // set aside; a queue entry without either is a worker spawning an engine against a book whose // bookkeeping does not exist. research/25 names the first of those explicitly ("a hold can leak // before the directory exists"), and the answer is not a compensating sweep but not splitting the // write in the first place. // // enqueue is handed the transaction so the queue's own insert joins it. It is a callback rather than // a second call because River owns its SQL and this package owns its own: neither reaches into the // other, and the atomicity is still real. func (s *Store) StartRun(ctx context.Context, in StartRunInput, journalOffset int64, enqueue func(context.Context, Tx, string) error) (StartedRun, error) { if in.CeilingChapters <= 0 { return StartedRun{}, fmt.Errorf("pgstore: a run needs a positive chapter ceiling, got %d", in.CeilingChapters) } out := StartedRun{JournalOffset: journalOffset} err := s.inTx(ctx, func(tx pgx.Tx) error { // The book first, before the hold — see lockBook for why the order is global and not local. // Here it also makes the race between two admissions of one book a queue instead of a // collision: the loser still gets ErrRunInFlight, it just gets it after waiting. if err := lockBook(ctx, tx, in.BookID); err != nil { return err } runID := newID("run") const insertRun = ` insert into runs (id, book_id, status, verify_bank, ceiling_chapters, started_at, revision) select $1, $2, 'translating', $3, $4, $5, b.revision + 1 from books b where b.id = $2 and b.owner_id = $6 returning id, book_id, revision, status, verify_bank, ceiling_chapters, coalesce(paused_reason, ''), started_at, finished_at` err := tx.QueryRow(ctx, insertRun, runID, in.BookID, in.VerifyBank, in.CeilingChapters, in.Now, in.UserID). Scan(&out.ID, &out.BookID, &out.Revision, &out.Status, &out.VerifyBank, &out.CeilingChapters, &out.PausedReason, &out.StartedAt, &out.FinishedAt) if errors.Is(err, pgx.ErrNoRows) { return ErrNoBook // the book is missing, or it is not this account's } if isUnique(err, "runs_one_live_per_book") { return ErrRunInFlight } if err != nil { return fmt.Errorf("pgstore: insert run: %w", err) } if err := tx.QueryRow(ctx, ` insert into run_attempts (run_id, attempt_no, started_at, last_offset) values ($1, 1, $2, $3) returning id`, out.ID, in.Now, journalOffset).Scan(&out.AttemptID); err != nil { return fmt.Errorf("pgstore: insert attempt: %w", err) } // The hold is taken here, BEFORE anything is spawned, and it is the enforcement half of the // money design rather than an accounting note. if err := holdTx(ctx, tx, in.UserID, in.BookID, engineRunKey(out.ID, 1), in.Ceiling, in.Now); err != nil { return err } if _, err := tx.Exec(ctx, ` update books set status = 'translating', revision = `+nextRevisionOfThisBooksLibrary+` where id = $1`, in.BookID); err != nil { return fmt.Errorf("pgstore: mark book translating: %w", err) } if enqueue != nil { return enqueue(ctx, tx, out.ID) } return nil }) if err != nil { return StartedRun{}, err } return out, nil } // engineRunKey is the reservation key of an attempt. // // ⚠ It is the PLATFORM's attempt identity, not the engine's `engine_run_id`: the engine mints its // own id per invocation and the platform only learns it from the handshake, which arrives after the // money has already been reserved. Reservations are keyed by attempt for exactly that reason — the // hold has to exist before there is anything to key it by on the engine's side. func engineRunKey(runID string, attempt int) string { return fmt.Sprintf("%s#%d", runID, attempt) } func isUnique(err error, constraint string) bool { var pg *pgconn.PgError return errors.As(err, &pg) && pg.Code == "23505" && pg.ConstraintName == constraint } // LiveRun is a run the reconciler has to make a decision about. type LiveRun struct { RunID string BookID string UserID string Workdir string AttemptID int64 AttemptNo int UnitName string EngineRunID string Position Position Quarantined bool VerifyBank bool // CeilingChapters is the run's whole budget, in the unit the user chose; Ceiling is what THIS // attempt was allowed to spend. They differ after a restart, which gets what is left. CeilingChapters int Ceiling money.MicroUSD // EngineBinary is the VERSIONED path this attempt was pinned to (unified backlog row 139). Read // back and USED — for the resume and for the repair channel — because a pin nothing reads is a // column, not a pin: the engine ships more often than a run finishes, and asking the CURRENT // binary about a book an older one is translating is a different question. EngineBinary string // SpendBaseline is what the BOOK had already cost when this attempt began. Nil means it was never // captured, which no current path produces (the spawn refuses without it). SpendBaseline *money.MicroUSD // CeilingArg is the limit a PREVIOUS claim of this attempt already handed the engine. Read back // because a retry must hand the same one: recomputing it from a counter that the first claim's own // engine has been moving gives that engine a second, larger limit and leaves the column that is // supposed to answer "what limit did that process have" describing neither. Zero means no claim // has been recorded (the column predates nothing else being able to tell). CeilingArg money.MicroUSD // PausedReason is what the stream already said about this run. A run that reported a ceiling halt // is `paused` however its process then ended (contract §BookStatus: never `failed`). PausedReason string // Status is the run's product status. The reconciler works on live runs, where it is // `translating`; the resume path reads a run that has already ended and decides by it. Status string // StopRequestedAt is when THIS platform asked the run to stop, and it is the only thing that can // tell a stop from a crash: the engine catches SIGTERM and exits 1, so the marker says // `exit-code/exited/1` for both (register row PD-152). Nil means nobody asked. StopRequestedAt *time.Time // StartedAt is the RUN's start; AttemptStartedAt is THIS attempt's. The reconciler's grace is // measured against the second: a restarted attempt inherits a start time hours old, and the // grace then expires before systemd has had a chance to create anything. StartedAt time.Time AttemptStartedAt time.Time } // Position mirrors ingest.Position without importing it: this package owns the columns, and the // dependency runs the other way. type Position struct { Offset int64 LastSeq int64 LastHash []byte } // ListLiveRuns returns every run that has not finished, with its live attempt. // // This is the reconciler's source of truth, together with the book's directory — NOT systemd // (research/25 §Опс). A transient unit does not survive a reboot and is unloaded the moment it // exits, so asking systemd "what is running" answers a different question than "what did this // platform promise a user". func (s *Store) ListLiveRuns(ctx context.Context) ([]LiveRun, error) { return s.queryRuns(ctx, ` join run_attempts a on a.run_id = r.id and a.ended_at is null left join reservations res on res.engine_run_id = r.id || '#' || a.attempt_no and res.state = 'open' where r.finished_at is null order by r.started_at`) } // runColumns is the reconciler's view of a run. Written once because the two queries that use it // differ only in which runs they select, and a scan list copied twice is a scan list that drifts. const runColumns = ` select r.id, r.book_id, b.owner_id, b.workdir, r.verify_bank, r.ceiling_chapters, coalesce(r.paused_reason, ''), r.started_at, r.status, r.stop_requested_at, a.id, a.attempt_no, coalesce(a.unit_name, ''), coalesce(a.engine_run_id, ''), a.last_offset, a.last_seq, a.last_line_sha256, a.quarantine_reason is not null, coalesce(res.amount_micro_usd, 0), a.engine_binary, a.spend_baseline_micro_usd, a.ceiling_arg_micro_usd, a.started_at from runs r join books b on b.id = r.book_id` func (s *Store) queryRuns(ctx context.Context, tail string, args ...any) ([]LiveRun, error) { rows, err := s.pool.Query(ctx, runColumns+tail, args...) if err != nil { return nil, fmt.Errorf("pgstore: list runs: %w", err) } defer rows.Close() var out []LiveRun for rows.Next() { var l LiveRun var ceiling, ceilingArg int64 var baseline *int64 if err := rows.Scan(&l.RunID, &l.BookID, &l.UserID, &l.Workdir, &l.VerifyBank, &l.CeilingChapters, &l.PausedReason, &l.StartedAt, &l.Status, &l.StopRequestedAt, &l.AttemptID, &l.AttemptNo, &l.UnitName, &l.EngineRunID, &l.Position.Offset, &l.Position.LastSeq, &l.Position.LastHash, &l.Quarantined, &ceiling, &l.EngineBinary, &baseline, &ceilingArg, &l.AttemptStartedAt); err != nil { return nil, fmt.Errorf("pgstore: scan run: %w", err) } l.Ceiling = money.MicroUSD(ceiling) l.CeilingArg = money.MicroUSD(ceilingArg) if baseline != nil { v := money.MicroUSD(*baseline) l.SpendBaseline = &v } out = append(out, l) } return out, rows.Err() } // RunSpent is what a run has actually been charged so far, across all its attempts. Settlements are // ledger rows keyed by attempt, so the sum is over the run's own key space and nothing else's. func (s *Store) RunSpent(ctx context.Context, runID string) (money.MicroUSD, error) { var v int64 err := s.pool.QueryRow(ctx, ` select coalesce(-sum(amount_micro_usd), 0) from credit_ledger where source = 'run_settle' and split_part(source_id, '#', 1) = $1`, runID).Scan(&v) if err != nil { return 0, fmt.Errorf("pgstore: read run spend: %w", err) } return money.MicroUSD(v), nil } // SpendBound is the UPPER bound on what an attempt can have cost: the smallest meter reading any // LATER attempt of the same book recorded before it started. // // It exists because settlement reads a lifetime counter of the BOOK at the moment it retries, and // that retry can happen after another run of the same book has already moved it. A settlement is // allowed to defer — the engine has to be asked and asking can fail — and a deferred one is not // blocked from being overtaken: the earlier run is finished, so nothing stops the account starting // another. Measured: a run that cost $0.10 was charged $2.10, the difference being what its // successor had spent by then, and the successor then paid that same amount again. // // A later attempt's baseline is exactly the right bound, because it was read BEFORE that attempt // added anything and AFTER this one had stopped. Nil means no later attempt has been spawned, and // then the counter has not been moved by anyone else. func (s *Store) SpendBound(ctx context.Context, bookID string, attemptID int64) (*money.MicroUSD, error) { var v *int64 err := s.pool.QueryRow(ctx, ` select min(a.spend_baseline_micro_usd) from run_attempts a join runs r on r.id = a.run_id where r.book_id = $1 and a.id > $2 and a.spend_baseline_micro_usd is not null`, bookID, attemptID).Scan(&v) if err != nil { return nil, fmt.Errorf("pgstore: read spend bound: %w", err) } if v == nil { return nil, nil } bound := money.MicroUSD(*v) return &bound, nil } // ErrRunNotLive is a run that cannot be stopped because it is already over. Distinct from ErrNoRun, // which is a run this account cannot see at all: the first is the contract's 409 and the second its // 404, and answering the wrong one either tells a stranger that a run exists or tells an owner that // theirs does not. var ErrRunNotLive = errors.New("pgstore: the run is not live") // ErrStopRequested is a restart refused because the run it would restart has been asked to stop. Not // an error of the caller: the reconciler ends the run instead. var ErrStopRequested = errors.New("pgstore: the run has been asked to stop") // RequestStop records that THIS platform asked a run to stop, and hands back the unit to ask. // // The order is the whole design (migration 00014): the intent is COMMITTED before systemd is // touched, so a platform that dies between the two still knows on its next sweep that the run it // finds ended was stopped on purpose — and, just as important, that it must not restart it. // // Idempotent by coalesce: pressing stop twice is one intent with the FIRST timestamp, because the // timestamp is evidence about which of the two events came first and a later one would erase that. // // The revision it answers with is the BOOK's — see ReadRun for why every run-carrying response uses // one counter. func (s *Store) RequestStop(ctx context.Context, userID, runID string, now time.Time) (Run, string, error) { const q = ` update runs r set stop_requested_at = coalesce(r.stop_requested_at, $3) from books b where r.id = $1 and b.id = r.book_id and b.owner_id = $2 and r.finished_at is null returning r.id, r.book_id, b.revision, r.status, r.verify_bank, r.ceiling_chapters, coalesce(r.paused_reason, ''), r.started_at, r.finished_at, coalesce((select a.unit_name from run_attempts a where a.run_id = r.id and a.ended_at is null order by a.attempt_no desc limit 1), '')` var out Run var unit string err := s.pool.QueryRow(ctx, q, runID, userID, now).Scan(&out.ID, &out.BookID, &out.Revision, &out.Status, &out.VerifyBank, &out.CeilingChapters, &out.PausedReason, &out.StartedAt, &out.FinishedAt, &unit) if errors.Is(err, pgx.ErrNoRows) { // Nothing matched, and the two reasons need different answers. Asked separately and only on // this path, so the ordinary stop stays one round trip. return Run{}, "", s.whyNotLive(ctx, userID, runID) } if err != nil { return Run{}, "", fmt.Errorf("pgstore: request stop: %w", err) } return out, unit, nil } func (s *Store) whyNotLive(ctx context.Context, userID, runID string) error { var visible bool if err := s.pool.QueryRow(ctx, ` select exists (select 1 from runs r join books b on b.id = r.book_id where r.id = $1 and b.owner_id = $2)`, runID, userID).Scan(&visible); err != nil { return fmt.Errorf("pgstore: read run: %w", err) } if visible { return ErrRunNotLive } return ErrNoRun } // ReadRunForResume loads a run and its LAST attempt, whether or not either is still live. // // The reconciler's own list is deliberately not reusable here: it selects the attempt that has not // ended, and every run this call is about has ended. What resume needs is the state the run stopped // in and the attempt whose money and journal position it stopped at. func (s *Store) ReadRunForResume(ctx context.Context, userID, runID string) (LiveRun, error) { rows, err := s.queryRuns(ctx, ` join run_attempts a on a.run_id = r.id and a.attempt_no = (select max(attempt_no) from run_attempts where run_id = r.id) left join reservations res on res.engine_run_id = r.id || '#' || a.attempt_no and res.state = 'open' where r.id = $1 and b.owner_id = $2`, runID, userID) if err != nil { return LiveRun{}, err } if len(rows) == 0 { return LiveRun{}, ErrNoRun } return rows[0], nil } // ReadRun is the run row as the contract projects it, read under the caller's ownership. // // ⚠ The revision is the BOOK's, not the `runs.revision` column, and that is the contract rather than // a shortcut: "the counter is PER BOOK — every book-scoped read and the id of every stream frame of // that book's run carry the same number" (§Revision), and a client MUST DROP a read whose revision is // below one it has applied. The book card already answers this way (httpapi getBook); a handle that // answered from the run's own column would hand the client a number below the card's and the client, // obeying the contract, would drop the answer to the button it just pressed. func (s *Store) ReadRun(ctx context.Context, userID, runID string) (Run, error) { const q = ` select r.id, r.book_id, b.revision, r.status, r.verify_bank, r.ceiling_chapters, coalesce(r.paused_reason, ''), r.started_at, r.finished_at from runs r join books b on b.id = r.book_id where r.id = $1 and b.owner_id = $2` var out Run err := s.pool.QueryRow(ctx, q, runID, userID).Scan(&out.ID, &out.BookID, &out.Revision, &out.Status, &out.VerifyBank, &out.CeilingChapters, &out.PausedReason, &out.StartedAt, &out.FinishedAt) if errors.Is(err, pgx.ErrNoRows) { return Run{}, ErrNoRun } if err != nil { return Run{}, fmt.Errorf("pgstore: read run: %w", err) } return out, nil } // RestartInput is one interrupted attempt being replaced. type RestartInput struct { RunID string AttemptID int64 UserID string BookID string // Ceiling is what is LEFT of the run's budget. A restart that reserved the full ceiling again // would let one run spend it twice. Ceiling money.MicroUSD Offset int64 // EngineBinary overrides the pinned path. Nil means INHERIT — which is the default, because a // resume is the same run continuing and row 139 lets another version in only on purpose. EngineBinary *string Now time.Time } // RestartRun closes an interrupted attempt and opens the next one, with its own hold, in ONE // transaction — the same reason StartRun is one transaction. // // The old attempt's reservation is NOT carried over: it was taken for a process that no longer // exists, and it is closed by the settlement that runs before this. func (s *Store) RestartRun(ctx context.Context, in RestartInput) (LiveRun, error) { var out LiveRun err := s.inTx(ctx, func(tx Tx) error { // Book first, as in every other transaction that touches both (lockBook). This one took the // attempt first and the materializer takes the book first, which is the pair that deadlocked. if err := lockBook(ctx, tx, in.BookID); err != nil { return err } // The RUN row second — the order lockBook writes down, which this transaction used to take // last, after the money. Taking it here also does the work below: what is read from the row is // read under the lock that will do the writing. var stopRequested, finished *time.Time if err := tx.QueryRow(ctx, `select stop_requested_at, finished_at from runs where id = $1 for update`, in.RunID). Scan(&stopRequested, &finished); err != nil { if errors.Is(err, pgx.ErrNoRows) { return ErrNoRun } return fmt.Errorf("pgstore: lock run: %w", err) } // A stop asked for on a run that is STILL LIVE outranks a restart, and the check belongs here // rather than in the caller's snapshot: the reconciler decides to restart, spends seconds // settling (a `tmctl status` call), and the user presses stop inside that window. Restarting // then cleared the fresh intent, opened a second attempt and took a new hold — the user had a // 202 for a stop that never happened and paid for the work they had just cancelled. // // On a FINISHED run the same column is history: a resume re-opens a run that was stopped, and // the intent belongs to the life that ended — it is cleared below with the rest of what said // the run was over. // // A stop arriving DURING this transaction is not lost either: it waits on this row lock and // lands on the re-opened run, which the next sweep then stops. if stopRequested != nil && finished == nil { return ErrStopRequested } // The previous attempt is closed if it is still open, and left exactly as it is if it is not. // Both callers arrive here: the reconciler replaces an attempt that was INTERRUPTED and is // still open, while a resume continues a run whose attempt already ended with a verdict of its // own — and overwriting that verdict would erase how the run the user stopped actually ended. // // What serializes two callers is no longer this row but the next one: `unique (run_id, // attempt_no)` lets exactly one of them insert attempt N+1, and the loser gets the same // ErrNoRun it always got. var attemptNo int if err := tx.QueryRow(ctx, ` update run_attempts set ended_at = coalesce(ended_at, $2), exit_result = coalesce(exit_result, 'interrupted') where id = $1 returning attempt_no`, in.AttemptID, in.Now).Scan(&attemptNo); err != nil { if errors.Is(err, pgx.ErrNoRows) { return ErrNoRun // the attempt is gone } return fmt.Errorf("pgstore: close interrupted attempt: %w", err) } next := attemptNo + 1 var id int64 var pinned string // The new attempt INHERITS the binary the run was pinned to (unified backlog row 139): the // engine ships more often than a run finishes, so resuming on whatever was deployed since is a // different program continuing someone else's work. Changing it is allowed, but only by saying // so — see runs.Config.AllowEngineVersionChange. if err := tx.QueryRow(ctx, ` insert into run_attempts (run_id, attempt_no, started_at, last_offset, engine_binary) select $1, $2, $3, $4, coalesce($5, prev.engine_binary, '') from run_attempts prev where prev.id = $6 returning id, engine_binary`, in.RunID, next, in.Now, in.Offset, in.EngineBinary, in.AttemptID). Scan(&id, &pinned); err != nil { if isUnique(err, "run_attempts_run_id_attempt_no_key") { return ErrNoRun // another caller opened this attempt first } return fmt.Errorf("pgstore: open next attempt: %w", err) } if err := holdTx(ctx, tx, in.UserID, in.BookID, engineRunKey(in.RunID, next), in.Ceiling, in.Now); err != nil { return err } // finished_at, settled_at and stop_requested_at are CLEARED, and each for its own reason. The // reconciler's restart works on a run where all three are already null, so there it changes // nothing; a RESUME re-opens a run that ended, and leaving them would mean a live run the // one-live-per-book index does not see, money marked resolved that this attempt has not spent // yet, and a stop request from the previous life that would classify this attempt's ending as // a stop nobody asked for. if _, err := tx.Exec(ctx, ` update runs set status = 'translating', paused_reason = null, finished_at = null, settled_at = null, stop_requested_at = null, revision = (select revision + 1 from books where id = runs.book_id) where id = $1`, in.RunID); err != nil { // Clearing finished_at puts the run back under the one-live-run-per-book index, and the // book may already have a NEWER live run — nothing stops an account starting one after it // stopped this one. That is a conflict and not a failure: the whole transaction rolls back, // so the hold taken three lines above is undone with it, and the caller gets the same error // a second admission would have got. if isUnique(err, "runs_one_live_per_book") { return ErrRunInFlight } return fmt.Errorf("pgstore: reopen run: %w", err) } if _, err := tx.Exec(ctx, ` update books set status = 'translating', revision = `+nextRevisionOfThisBooksLibrary+` where id = $1`, in.BookID); err != nil { return fmt.Errorf("pgstore: mark book translating: %w", err) } out = LiveRun{RunID: in.RunID, BookID: in.BookID, UserID: in.UserID, AttemptID: id, AttemptNo: next, Ceiling: in.Ceiling, StartedAt: in.Now, AttemptStartedAt: in.Now, EngineBinary: pinned, Position: Position{Offset: in.Offset}} return nil }) if err != nil { return LiveRun{}, err } // The fields the caller needs to spawn but this transaction did not read. const q = `select b.workdir, r.verify_bank, r.ceiling_chapters from runs r join books b on b.id = r.book_id where r.id = $1` if err := s.pool.QueryRow(ctx, q, in.RunID).Scan(&out.Workdir, &out.VerifyBank, &out.CeilingChapters); err != nil { return LiveRun{}, fmt.Errorf("pgstore: read restarted run: %w", err) } return out, nil } // ReleaseSpawnClaim gives an attempt back after a unit could NOT be created, so the next sweep // retries this attempt instead of reading a recorded-but-absent unit as an interrupted run. func (s *Store) ReleaseSpawnClaim(ctx context.Context, attemptID int64) error { _, err := s.pool.Exec(ctx, `update run_attempts set unit_name = null where id = $1`, attemptID) if err != nil { return fmt.Errorf("pgstore: release spawn claim: %w", err) } return nil } // AttemptReservationOpen reports whether an attempt's money is still reserved. The restart path asks // before it opens a SECOND reservation: settling can legitimately fail (the engine's figure could not // be read), and proceeding then holds the ceiling twice and strands the first hold where no sweep // looks for it again. func (s *Store) AttemptReservationOpen(ctx context.Context, runID string, attempt int) (bool, error) { var open bool err := s.pool.QueryRow(ctx, ` select exists (select 1 from reservations where engine_run_id = $1 and state = 'open')`, engineRunKey(runID, attempt)).Scan(&open) if err != nil { return false, fmt.Errorf("pgstore: read reservation state: %w", err) } return open, nil } // PauseRun records a run that cannot go on, without pretending it failed. func (s *Store) PauseRun(ctx context.Context, runID string, attemptID int64, reason string, now time.Time) error { if reason != PausedCreditExhausted { return fmt.Errorf("pgstore: %q is not a pause reason", reason) } return s.inTx(ctx, func(tx Tx) error { // Book first, as everywhere else that touches both (see FinishRun). var bookID string if err := tx.QueryRow(ctx, `select id from books where id = (select book_id from runs where id = $1) for update`, runID).Scan(&bookID); err != nil { if errors.Is(err, pgx.ErrNoRows) { return nil } return fmt.Errorf("pgstore: lock book: %w", err) } // A stop the user asked for outranks a pause, and the check belongs inside this TRANSACTION, // under the same locks: the reconciler decides to pause after settling — seconds of a // `tmctl status` call — and a stop landing inside that window would otherwise be answered with // `paused/credit_exhausted`, which // says the money ran out when what happened is that its owner stopped it. var stopRequested *time.Time if err := tx.QueryRow(ctx, `select stop_requested_at from runs where id = $1 for update`, runID).Scan(&stopRequested); err != nil { if errors.Is(err, pgx.ErrNoRows) { return nil } return fmt.Errorf("pgstore: lock run: %w", err) } if stopRequested != nil { return ErrStopRequested } // The attempt being closed must still be a LIVE attempt OF THIS RUN — the third path to carry // the guard `FinishRun` (PD-181) and `FinishUnspawnedStop` (FP5-2) already have, and it was // missing here. Without it a pass holding an old snapshot pauses a run whose attempt has since // been restarted: the run reads `finished`, the SECOND attempt's hold stays open and falls out // of both `ListLiveRuns` (the run is finished) and `UnsettledRuns` (the attempt is not), and an // engine keeps spending under a run its owner is told ran out of money. Found by cross-family // review of the acceptance dofix. if err := tx.QueryRow(ctx, ` update runs set status = 'paused', paused_reason = $2, finished_at = $3, revision = (select revision + 1 from books where id = runs.book_id) where id = $1 and finished_at is null and exists (select 1 from run_attempts a where a.id = $4 and a.run_id = runs.id and a.ended_at is null) returning book_id`, runID, reason, now, attemptID).Scan(&bookID); err != nil { if errors.Is(err, pgx.ErrNoRows) { return nil } return fmt.Errorf("pgstore: pause run: %w", err) } if _, err := tx.Exec(ctx, ` update run_attempts set ended_at = coalesce(ended_at, $3) where id = $1 and run_id = $2`, attemptID, runID, now); err != nil { return fmt.Errorf("pgstore: end attempt: %w", err) } if _, err := tx.Exec(ctx, ` update books set status = 'paused', revision = `+nextRevisionOfThisBooksLibrary+` where id = $1`, bookID); err != nil { return fmt.Errorf("pgstore: pause book: %w", err) } return nil }) } // RunForSpawn is what the worker needs to spawn an attempt. type RunForSpawn struct { LiveRun // AlreadySpawned is a unit name that was recorded for this attempt. The worker must NOT spawn a // second engine for it: a queue job can be retried after a platform restart, and the run it // refers to may well still be running. AlreadySpawned bool } // ReadRunForSpawn loads the live attempt of a run. func (s *Store) ReadRunForSpawn(ctx context.Context, runID string) (RunForSpawn, error) { live, err := s.ListLiveRuns(ctx) if err != nil { return RunForSpawn{}, err } for _, l := range live { if l.RunID == runID { return RunForSpawn{LiveRun: l, AlreadySpawned: l.UnitName != ""}, nil } } return RunForSpawn{}, ErrNoRun } // SpawnRecord is what is written down about an attempt just before its unit is created. type SpawnRecord struct { AttemptID int64 Unit string // Binary is the versioned engine path this attempt is pinned to (unified backlog row 139). Binary string // Ceiling is the attempt's own budget: the increment the user bought and the amount held. Ceiling money.MicroUSD // CeilingArg is what the engine is actually told, which is the book's cumulative cap and // therefore a different number from the second run of a book onwards (D39.122). CeilingArg money.MicroUSD // Baseline is where the book's lifetime meter stood before this attempt added to it. Baseline money.MicroUSD } // RecordSpawn claims the right to start this attempt, and writes down what is about to run: the // unit, the binary version it is pinned to, the budget it carries and the ceiling it is given. // Written BEFORE the unit is created, so a crash between the two leaves a record to reconcile rather // than an unattributable process. // // claimed is false when the attempt already has a unit name, when the attempt has ENDED, or when its // run is over. It is a COMPARE-AND-SET rather than a plain update because two callers legitimately // reach here at once — the queue worker that was handed the run and the reconciler that found it // unspawned — and the loser must not start a second engine. systemd would refuse the duplicate NAME, // so that accident was survivable; surviving by someone else's uniqueness rule is not the same as // being correct, and the day a resume changes the naming it stops holding. // // ⚠ The three conditions BESIDES the unit name are money, and they were not here until a run could be // stopped before it ever spawned. The window: a worker sits inside `bookMeter` for the seconds a // `tmctl status` takes, the user stops the run, the sweep ends it and gives the whole hold back // (nothing was spawned) — and the worker then wakes up and creates a unit for a run that is finished // and settled. That engine would spend against its own book cap with NO open reservation, and // nothing would ever look at it: the reconciler lists runs by `finished_at is null` and settlements // by an open reservation, so it is in neither list. The stop INTENT is asked about one step earlier // for the same reason: it lands while the worker is inside that same `bookMeter`, before any sweep // could have finished the run, and a claim granted then starts an engine for work its owner // cancelled before it began. All three are asked in the statement that claims, because a check made // before it is a check with a window after it. func (s *Store) RecordSpawn(ctx context.Context, r SpawnRecord) (claimed bool, err error) { // What a PREVIOUS claim of this attempt decided is kept. The claim is given back when the unit // could not be created (ReleaseSpawnClaim), and "could not be created" is not the same as "was not // created": a systemd-run that was killed after it had already asked for the unit reports a failure // and leaves an engine running. The next claim would then overwrite the baseline with a meter that // engine has been moving, and the attempt would be billed for the difference from a figure that // already includes its own work — an underpayment nothing later looks for. Where the unit really // was not created the two values are identical, so keeping the first costs nothing. tag, err := s.pool.Exec(ctx, ` update run_attempts set unit_name = $2, engine_binary = $3, ceiling_micro_usd = $4, ceiling_arg_micro_usd = case when spend_baseline_micro_usd is null then $5 else ceiling_arg_micro_usd end, spend_baseline_micro_usd = coalesce(spend_baseline_micro_usd, $6) where id = $1 and unit_name is null and ended_at is null and exists (select 1 from runs r where r.id = run_attempts.run_id and r.finished_at is null and r.stop_requested_at is null)`, r.AttemptID, r.Unit, r.Binary, int64(r.Ceiling), int64(r.CeilingArg), int64(r.Baseline)) if err != nil { return false, fmt.Errorf("pgstore: record spawn: %w", err) } return tag.RowsAffected() == 1, nil } // SaveCursor persists the tailer's position when nothing was materialized — a re-read of lines the // cursor already covers still moves the byte hint, and losing that means re-reading them forever. func (s *Store) SaveCursor(ctx context.Context, attemptID int64, p Position) error { _, err := s.pool.Exec(ctx, ` update run_attempts set last_offset = $2 where id = $1 and last_offset < $2`, attemptID, p.Offset) if err != nil { return fmt.Errorf("pgstore: save cursor: %w", err) } return nil } // Quarantine stops materializing an attempt without touching the run. // // The engine is NOT stopped: it is spending money the account has already reserved, and our // inability to read its journal is not a reason to throw that away. What stops is the projection — // after this the run's state is only as fresh as the resync channel makes it, and the reason says // so out loud rather than leaving a screen that quietly stopped moving. func (s *Store) Quarantine(ctx context.Context, attemptID int64, reason string) error { _, err := s.pool.Exec(ctx, `update run_attempts set quarantine_reason = $2 where id = $1 and quarantine_reason is null`, attemptID, reason) if err != nil { return fmt.Errorf("pgstore: quarantine attempt: %w", err) } return nil }