1687 lines
93 KiB
Go
1687 lines
93 KiB
Go
package runs
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"path/filepath"
|
|
"time"
|
|
|
|
"textmachine/platform/internal/ingest"
|
|
"textmachine/platform/internal/money"
|
|
"textmachine/platform/internal/pgstore"
|
|
"textmachine/platform/internal/runner"
|
|
)
|
|
|
|
// spawnGrace is how long after an attempt was admitted the reconciler still believes systemd simply
|
|
// has not got to it. Under it, "no unit and no marker" means "not started yet"; over it, the same
|
|
// two facts mean the unit died without being able to say so — which is what a reboot looks like,
|
|
// since a transient unit does not survive one and its ExecStopPost never runs.
|
|
const spawnGrace = 60 * time.Second
|
|
|
|
// Sweep reconciles every run the database believes is live, then settles the money of every run
|
|
// that finished without it.
|
|
//
|
|
// It runs on a ticker AND at boot, and the boot pass is not a special case: the same reading of the
|
|
// same three sources — Postgres, the book's journal, the exit marker — restarts what a reboot
|
|
// interrupted (unified backlog row 138). systemd is asked only "is this unit still there", and its
|
|
// answer is evidence rather than truth (research/25 §Опс).
|
|
//
|
|
// One run's failure never stops the sweep: these are independent runs belonging to independent
|
|
// accounts, and a book whose directory an operator moved must not freeze everyone else's progress.
|
|
// ⚠ THE TWO PHASES GET THEIR OWN SHARES OF THE PASS, and that is not tidiness — it is the whole of
|
|
// what stopped the settlement of an entire installation. The pass is bounded (120 s) and each run
|
|
// inside it is bounded (60 s), so TWO runs whose engine hangs consumed the pass exactly, `ctx.Err()`
|
|
// returned out of the loop, and `UnsettledRuns` — the ONLY retry a deferred settlement has — was not
|
|
// reached at all. Not occasionally: the list is ordered oldest-first and a wedged run is by
|
|
// construction the oldest, so it happened on every pass, for everyone, until somebody noticed. Money
|
|
// froze for accounts that had nothing to do with the wedged books.
|
|
//
|
|
// Settlement gets the SECOND half rather than the first because it is the cheaper work and the one
|
|
// with a deadline that matters to a user: a hold that is still reserved is a book the account cannot
|
|
// start another run on.
|
|
//
|
|
// ⚠ BOTH PHASES DEFER. Splitting the pass stops one phase starving the other; it does nothing about
|
|
// one item starving the rest of its own phase, and the settlement list is ordered oldest-first too.
|
|
func (s *Service) Sweep(ctx context.Context) error {
|
|
return errors.Join(s.reconcilePhase(ctx), s.settlePhase(ctx))
|
|
}
|
|
|
|
// reconcilePhase works through the runs that are DUE, under its own share of the pass.
|
|
func (s *Service) reconcilePhase(ctx context.Context) error {
|
|
phase, cancel := s.phaseBudget(ctx)
|
|
defer cancel()
|
|
live, err := s.Store.RunsToReconcile(phase, s.now())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
for i, l := range live {
|
|
if over(phase) {
|
|
return s.ranOut(ctx, phase, "the run sweep", "live runs", i, len(live))
|
|
}
|
|
s.reconcileOne(phase, l)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// ranOut is what a phase says when the pass ended under it.
|
|
//
|
|
// REPORTED as a deadline rather than swallowed: `sweep_unfinished_total` is the one signal that says
|
|
// starvation is happening (register row PD-169) and the caller raises it by testing for this error.
|
|
// A CANCELLED pass is a different fact — the daemon going down — and reporting it as starvation both
|
|
// raised that counter on every restart and wrote unfinished items a failure they had not had.
|
|
//
|
|
// Items left over are not deferred: they were never touched, so they simply wait for the next pass.
|
|
func (s *Service) ranOut(ctx, phase context.Context, sweep, items string, done, total int) error {
|
|
if !errors.Is(phase.Err(), context.DeadlineExceeded) {
|
|
return nil // the daemon is going down, not a pass that could not keep up
|
|
}
|
|
s.log().WarnContext(ctx, "a sweep ran out of its share of the pass", "sweep", sweep, "unreached", total-done)
|
|
return fmt.Errorf("runs: %s reached %d of %d %s: %w", sweep, done, total, items, context.DeadlineExceeded)
|
|
}
|
|
|
|
// reconcileOne runs one item under its own budget and records what came of it.
|
|
//
|
|
// A failure or a timeout DEFERS this attempt, and the deferral is the answer to head-of-line
|
|
// starvation: the list is oldest-first, so an attempt that can never succeed held the head of it on
|
|
// every pass and the runs behind it were never reached. Deferring takes it out of the head instead
|
|
// of changing the ordering, because oldest-first is the right order for everything that is only slow.
|
|
func (s *Service) reconcileOne(ctx context.Context, l pgstore.LiveRun) {
|
|
item, done := context.WithTimeout(ctx, s.runBudget())
|
|
established, err := s.reconcile(item, l)
|
|
// ⚠ SPENDING the budget counts as a failure even when nothing reported one, and that is not a
|
|
// detail: the commonest wedge — a `tmctl status` that never answers on a live run — returns NO
|
|
// error at all. `maybeResync` swallows it on purpose ("a status call that fails is not a run that
|
|
// failed"), correctly, so keying the deferral on the error alone leaves exactly the case this
|
|
// whole mechanism exists for at the head of the list forever.
|
|
//
|
|
// ⚠ RUNNING OUT OF TIME IS THE FAILURE, without qualification, and the two attempts to qualify it
|
|
// were both wrong in the same way. Both asked whether the PARENT still had time — but the item's
|
|
// context is a child created later, so the phase's deadline is never the later of the two, and at
|
|
// the shipped defaults (a 2-minute pass, so a 60-second phase, against a 60-second run budget)
|
|
// they are the same instant. The qualifier was therefore false exactly when the item had hung: the
|
|
// other branch ran and recorded a WEDGED run as successfully reconciled, clearing whatever count
|
|
// it had. Deferral, the stalled gauge and the operator's handle were unreachable on any default
|
|
// deployment while every test of them passed, because every one of those tests set a run budget
|
|
// far below the pass. Found by a reviewer's pin written at the shipped ratio.
|
|
//
|
|
// What the qualifier was FOR — not punishing a run that was merely last in a busy tick — is given
|
|
// up deliberately, on the asymmetry: a deferral nobody deserved costs one minute and is erased by
|
|
// the first pass that succeeds, while one that is missed costs forever. And a run cut short on
|
|
// five consecutive passes is not a false alarm either way: that is starvation, which is the thing
|
|
// an operator is meant to hear about.
|
|
// The item's own clock, not merely a context that is unusable: a shutdown cancels every item in
|
|
// flight, and reading that as "this run spent its budget" writes failures a restart invented.
|
|
overran := errors.Is(item.Err(), context.DeadlineExceeded)
|
|
done()
|
|
// Detached: the write that RECORDS a timeout must not run on the context that just expired, or
|
|
// the failure is not counted and the attempt is at the head of the list again next pass.
|
|
c, cancel := context.WithTimeout(context.WithoutCancel(ctx), recordBudget)
|
|
defer cancel()
|
|
if err == nil && !overran {
|
|
if !established {
|
|
// ⚠ CLEARED BY EVIDENCE, never by the absence of it. Most passes over a live run prove
|
|
// nothing — the expensive question is asked once every ResyncEvery and skipped while the
|
|
// stream is moving — so a run whose `tmctl status` hangs failed, was deferred, came back to
|
|
// a pass that asked the engine nothing, and had its count reset by that silence. It
|
|
// oscillated 1,0,1,0: StalledAfter was unreachable and no operator was ever told.
|
|
return
|
|
}
|
|
if err := s.Store.ClearRunDeferral(c, l.AttemptID); err != nil {
|
|
s.log().ErrorContext(c, "the run's deferral could not be cleared", "run", l.RunID, "err", err)
|
|
}
|
|
return
|
|
}
|
|
reason := "the run took its whole budget without finishing"
|
|
blocked := errors.Is(err, errSettlementBlocked)
|
|
switch {
|
|
case blocked:
|
|
// ONE sentence for one fact, whichever phase reports it: the operator's table prints this
|
|
// under LAST ERROR, and «the settlement could not be computed» read from two phases in two
|
|
// different wordings would look like two different problems.
|
|
reason = settleReason(settlementBlocked, nil, false)
|
|
s.log().ErrorContext(ctx, "the run's settlement is blocked, so it cannot be restarted; it is deferred",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
case err != nil:
|
|
reason = err.Error()
|
|
s.log().ErrorContext(ctx, "run could not be reconciled", "run", l.RunID, "err", err)
|
|
default:
|
|
s.log().WarnContext(ctx, "the run took its whole budget without finishing; it is deferred",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
}
|
|
// ⚠ THE SETTLEMENT'S CAP WHEN THE ATTEMPT IS ALREADY OVER, and this closes a hole acceptance
|
|
// measured at thirty minutes. `finish`, `finishStopped` and `restart` all END the attempt and
|
|
// then settle, so an error surfacing here can belong to an attempt that has already left this
|
|
// phase for the settlement list — and deferring it with the LIVE backoff (capped at half an
|
|
// hour) put the user's own resume behind that half hour, which is precisely what
|
|
// `settlementBackoffCap` was shortened to five minutes to prevent. The read costs one indexed
|
|
// lookup and only on a failure.
|
|
delay := backoff(l.ReconcileFailures + 1)
|
|
if blocked {
|
|
// The SETTLEMENT's schedule, not the live phase's, because it is the settlement that is stuck
|
|
// and the settlement's deferral is the USER's resume gate (settlementBackoffCap's own
|
|
// argument). Half an hour of live backoff would put their resume behind it for a blockage
|
|
// they can do nothing about.
|
|
delay = settlementDelay(l.ReconcileFailures)
|
|
}
|
|
switch ended, qerr := s.Store.AttemptEnded(c, l.AttemptID); {
|
|
case qerr != nil:
|
|
s.log().ErrorContext(c, "could not tell whether the attempt had ended; deferring as a live run",
|
|
"run", l.RunID, "err", qerr)
|
|
case ended:
|
|
delay = min(delay, settlementBackoffCap)
|
|
}
|
|
s.deferItem(c, l, reason, delay)
|
|
}
|
|
|
|
// deferItem takes one item out of the head of its list and says so when it has been there too long.
|
|
//
|
|
// Shared by both phases: the same fact about the same row, and the two phases work on disjoint sets
|
|
// of it (attempts that have not ended, attempts that have), so one mechanism serves both.
|
|
//
|
|
// ⚠ The DELAY is the caller's, because the two phases pay differently for waiting: a live run that
|
|
// is deferred costs the platform a slower repair, while a deferred SETTLEMENT is also the user's
|
|
// resume gate (settleOne, settlementDelay).
|
|
func (s *Service) deferItem(ctx context.Context, l pgstore.LiveRun, reason string, after time.Duration) {
|
|
failures, err := s.Store.DeferRun(ctx, l.AttemptID, s.now().Add(after), reason)
|
|
if err != nil {
|
|
s.log().ErrorContext(ctx, "the run could not be deferred; it will be retried at the head of the next pass",
|
|
"run", l.RunID, "err", err)
|
|
return
|
|
}
|
|
if failures == StalledAfter {
|
|
// Said ONCE, at the crossing, and not on every pass afterwards: this is the line an operator
|
|
// is meant to act on, and a line repeated every fifteen seconds is one they filter out.
|
|
s.log().ErrorContext(ctx, "a run has failed to reconcile enough times to be called stalled; it is now retried rarely and needs an operator (tmplatformctl runs)",
|
|
"run", l.RunID, "attempt", l.AttemptNo, "failures", failures)
|
|
}
|
|
}
|
|
|
|
// settlePhase closes the money of runs that ended without it. It runs whatever the phase above did.
|
|
func (s *Service) settlePhase(ctx context.Context) error {
|
|
// Nothing left of the pass at all: the reconciliation phase used it. Not reported as unfinished
|
|
// HERE — the phase above already reported it, and one pass counted twice would read as two.
|
|
if over(ctx) {
|
|
return nil
|
|
}
|
|
unsettled, err := s.Store.UnsettledRuns(ctx, s.now())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
for i, u := range unsettled {
|
|
if over(ctx) {
|
|
return s.ranOut(ctx, ctx, "the settlement sweep", "finished runs", i, len(unsettled))
|
|
}
|
|
s.settleOne(ctx, u)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// settleOne closes the money of one finished attempt under its own budget, and takes it out of the
|
|
// head of the list when it could not.
|
|
//
|
|
// ⚠ WHAT COUNTS AS A FAILURE IS THE VERDICT, not the cost, and reading the cost instead was PD-384.
|
|
// This used to defer only on a spent budget, because `settle` answers nil when the engine's figure
|
|
// cannot be read — and that is right as an ERROR, an unreadable spend is a settlement to retry. But
|
|
// a settlement that cannot be computed can be permanent: the engine binary of a pinned build removed
|
|
// at a rollout, an attempt from before the baseline column existed, a store error on the bound this
|
|
// settlement needs. Each is CHEAP, so `overran` was false, so nothing counted — the counter stayed at
|
|
// zero, `reconcile_after` stayed null, the threshold was unreachable, the gauge and
|
|
// `tmplatformctl runs --stalled` stayed empty, and the hold stayed frozen with nobody told.
|
|
// Measured: five passes of one uncomputable run gave five engine calls, `reconcile_failures` 0 and
|
|
// `StalledRuns(5)` 0.
|
|
//
|
|
// Deferring it fixes the second half of the same row: `settle` calls the engine's `status` on EVERY
|
|
// pass with nothing to rate-limit it, while its neighbour `maybeResync` grew `dueForResync` for
|
|
// exactly that cost. The backoff IS the rate limit.
|
|
//
|
|
// ⚠ AND THE FIRST FAILURE IS NOT DEFERRED, because the delay is NOT free and saying it was would be
|
|
// the comment lying about the code. An open reservation is also the USER's resume gate — `reopen`
|
|
// refuses to start the next attempt while the previous one's hold is open — so every minute this
|
|
// waits is a minute their `POST /runs/{id}/resume` is answered 409. The commonest cause of a single
|
|
// blocked settlement is transient — the engine binary momentarily unavailable mid-rollout, or a
|
|
// store error on the bound this settlement needs — and it used to clear on the next pass, fifteen
|
|
// seconds later. (⚠ NOT "the project file still locked by the exiting process", which an earlier
|
|
// draft of this comment claimed: `tmctl status` opens the project READ-ONLY and takes no exclusive
|
|
// lock at all, precisely so an operator is not locked out during a run. A comment naming a cause the
|
|
// code cannot produce is how a later session calibrates a number against nothing.) So the
|
|
// first failure is COUNTED and made due again immediately, which keeps that fifteen seconds; the
|
|
// backoff starts from the second, where "transient" has stopped being the likely explanation. The
|
|
// threshold is still reached, one pass later than a pure backoff would reach it.
|
|
func (s *Service) settleOne(ctx context.Context, u pgstore.LiveRun) {
|
|
item, done := context.WithTimeout(ctx, s.runBudget())
|
|
verdict, err := s.settle(item, u)
|
|
overran := errors.Is(item.Err(), context.DeadlineExceeded)
|
|
done()
|
|
if err != nil {
|
|
s.log().ErrorContext(ctx, "run could not be settled", "run", u.RunID, "err", err)
|
|
}
|
|
// ⚠ THE OVERRUN KEEPS THE ORDINARY BACKOFF and does not get the fast first retry below. The
|
|
// grace is argued from a CHEAP transient failure — an engine momentarily unavailable, a store
|
|
// error — that used to clear on the very next pass. A settlement that ate the whole run budget
|
|
// was never cheap and never cleared next pass, so handing it a zero delay would put it back at
|
|
// the head of the list immediately and cost the phase its budget again. This was a real leak in
|
|
// this pack's first shape of the fix, found at acceptance: the zero was written for one branch
|
|
// and reached three.
|
|
// ⚠ A SHUTDOWN IS NOT A FAILURE, and this branch was missing until acceptance found it. Stopping
|
|
// the daemon cancels every item in flight, so `settle` comes back with `context.Canceled` while
|
|
// `overran` is false — and everything below then wrote a failure and a deferral on a context
|
|
// deliberately detached from the cancellation, which commits. Measured: an ordinary pass left
|
|
// `failures=1`, the same pass under a shutdown left `failures=2` and a two-minute deferral. That
|
|
// is a count a RESTART invented, and the reconciliation phase next door refuses it in exactly
|
|
// these words ("writes failures a restart invented") — this phase had lost the same guard.
|
|
if ctx.Err() != nil {
|
|
return
|
|
}
|
|
delay := settlementDelay(u.ReconcileFailures)
|
|
switch {
|
|
case overran:
|
|
delay = backoff(u.ReconcileFailures + 1)
|
|
s.log().WarnContext(ctx, "the settlement took its whole budget without finishing; it is deferred",
|
|
"run", u.RunID, "attempt", u.AttemptNo)
|
|
case err == nil && verdict == settlementClosed:
|
|
// Nothing is cleared here: an attempt whose money closes leaves this list for good, so the
|
|
// count it carried dies with its place in it.
|
|
return
|
|
case err == nil && verdict == settlementRaced:
|
|
// Not a failure and not a settlement: the sweep's snapshot was older than the attempt's own
|
|
// row, and the next pass reads the spawn and settles the ordinary way. Counting it would put
|
|
// a race — which resolves by itself, on the very next tick — on the counter an operator acts
|
|
// on, and five of them would raise an alarm about a system that is working.
|
|
return
|
|
}
|
|
c, cancel := context.WithTimeout(context.WithoutCancel(ctx), recordBudget)
|
|
defer cancel()
|
|
s.deferItem(c, u, settleReason(verdict, err, overran), delay)
|
|
}
|
|
|
|
// settlementDelay is how long a settlement that could not be COMPUTED waits before the sweep looks
|
|
// at its money again. An overrun is not its business: see settleOne.
|
|
//
|
|
// Zero on the FIRST failure — see settleOne for why that minute is not free — and the ordinary
|
|
// backoff afterwards. A zero delay is not "no deferral": the failure is still counted, and the
|
|
// worklist admits the row again on the next pass (`reconcile_after <= now`), which is exactly where
|
|
// it was before this pack.
|
|
func settlementDelay(failures int) time.Duration {
|
|
if failures == 0 {
|
|
return 0
|
|
}
|
|
return min(backoff(failures+1), settlementBackoffCap)
|
|
}
|
|
|
|
// settlementBackoffCap is where the settlement phase stops doubling, and it is SHORTER than the
|
|
// reconciliation phase's thirty minutes on purpose.
|
|
//
|
|
// The two deferrals cost different people. A deferred live run costs the platform a slower repair; a
|
|
// deferred SETTLEMENT costs the USER, because the open reservation is what `reopen` refuses to start
|
|
// the next attempt over — so the wait is the wait before their resume stops answering 409. Half an
|
|
// hour of that after a fault that healed in seconds is not a rate limit, it is an outage of our own
|
|
// making. Five minutes still bounds the engine call to one per book per five minutes, which is the
|
|
// cost half of what PD-384 asked for.
|
|
//
|
|
// ⚠ It binds BOTH phases, and that is not tidiness. An attempt can end and fail to settle inside a
|
|
// single pass of the RECONCILIATION phase (`finish` settles opportunistically), and that phase's own
|
|
// backoff caps at thirty minutes — so without applying this cap there too, the five-minute promise
|
|
// held only for settlements the settlement phase happened to meet first. Measured at acceptance:
|
|
// thirty minutes, with `Resume` answering 409 throughout.
|
|
const settlementBackoffCap = 5 * time.Minute
|
|
|
|
// settleReason is what goes on the attempt's row, because that string is what the operator's table
|
|
// prints under LAST ERROR and it is the only account of why the money is not moving.
|
|
func settleReason(verdict settleVerdict, err error, overran bool) string {
|
|
switch {
|
|
case overran:
|
|
return "the settlement took its whole budget without finishing"
|
|
case err != nil:
|
|
return "the settlement failed: " + err.Error()
|
|
}
|
|
// Everything left is `settlementBlocked`: settleOne returns without deferring for the other two
|
|
// verdicts, so there is no fourth sentence to write. An earlier shape had one, and a LAST ERROR
|
|
// string no state can produce is a sentence an operator could be shown and nobody could explain.
|
|
return "the settlement could not be computed"
|
|
}
|
|
|
|
// errSettlementBlocked is a RESTART that could not be attempted because the money of the previous
|
|
// attempt could not be closed. It is an error rather than a silent return, and that is the whole of
|
|
// register row PD-424.
|
|
//
|
|
// The chain it ends: a unit vanishes without a marker → `restart` → `settle` answers
|
|
// `settlementBlocked` WITHOUT an error → the old code threw the verdict away, walked into `reopen`,
|
|
// which refuses to open an attempt over an unclosed hold and answers `deferred` → `case deferred:
|
|
// return nil` → `reconcileOne` reads a successful pass and CLEARS the deferral. Every surface stayed
|
|
// clean: `reconcile_failures` never left 0, `reconcile_after` stayed NULL, `StalledRuns` was empty,
|
|
// the stalled gauge read 0 — while the hold stayed frozen and the engine was poked every fifteen
|
|
// seconds forever. Nothing but the user pressing Stop could end it, and nobody told them.
|
|
//
|
|
// The question the row said had to be answered first — what `deferred` MEANS for the counter — is
|
|
// answered here and only here: a settlement that cannot be COMPUTED is a failure of the phase that
|
|
// owns it, and it is counted by that phase. `reopen`'s own `deferred` verdict is left exactly as it
|
|
// was, because after this it can only be reached on a `settlementRaced` whose reservation is briefly
|
|
// open — self-correcting on the next pass, and the settlement phase argues in as many words that a
|
|
// race must not reach an operator's counter.
|
|
var errSettlementBlocked = errors.New("runs: the attempt's settlement could not be computed, so the run cannot be restarted")
|
|
|
|
// settleVerdict is what one pass over an attempt's money produced. Three states and not a boolean,
|
|
// because the middle one must NOT reach the failure counter: see settleOne.
|
|
type settleVerdict int
|
|
|
|
const (
|
|
// settlementClosed: the money is resolved and the attempt leaves the worklist for good.
|
|
settlementClosed settleVerdict = iota
|
|
// settlementRaced: the sweep's snapshot predated a spawn. Self-correcting on the next pass.
|
|
settlementRaced
|
|
// settlementBlocked: it could not be computed, and nothing here knows whether it ever will be.
|
|
settlementBlocked
|
|
)
|
|
|
|
// phaseBudget is the reconciliation phase's share of the pass: half of what is left, so the
|
|
// settlement phase cannot be starved by it however slow the engine is.
|
|
//
|
|
// Half rather than a constant, because the pass's own budget is an operator's number now
|
|
// (TM_PLATFORM_SWEEP_BUDGET) and a constant here would silently stop being a share of it.
|
|
// ⚠ A TIMEOUT and not a deadline computed from `s.now()`. This service's clock is injectable and a
|
|
// test freezes it, while a context expires on the real one — so an absolute deadline built from the
|
|
// two gave the phase whatever the offset between them happened to be. Caught by the pin below, which
|
|
// measured a pass twice as long as it had asked for.
|
|
func (s *Service) phaseBudget(ctx context.Context) (context.Context, context.CancelFunc) {
|
|
deadline, ok := ctx.Deadline()
|
|
if !ok {
|
|
return context.WithCancel(ctx)
|
|
}
|
|
return context.WithTimeout(ctx, time.Until(deadline)/2)
|
|
}
|
|
|
|
// over reports a context whose time is up, without producing an error to discard.
|
|
func over(ctx context.Context) bool {
|
|
select {
|
|
case <-ctx.Done():
|
|
return true
|
|
default:
|
|
return false
|
|
}
|
|
}
|
|
|
|
// recordBudget is what WRITING DOWN the outcome of one item gets, on a context of its own. Short: it
|
|
// is one statement against a database this process is already connected to.
|
|
const recordBudget = 10 * time.Second
|
|
|
|
// StalledAfter is how many consecutive failures make an attempt an operator's problem rather than a
|
|
// slow one. Four backoffs at the shape below is about a quarter of an hour of trying, which is long
|
|
// enough that a restart of Postgres or a busy host clears by itself and short enough that a person
|
|
// hears about a wedge on the same shift.
|
|
const StalledAfter = 5
|
|
|
|
// backoff is how long an attempt waits before the sweep looks at it again.
|
|
//
|
|
// Doubling from one minute and capped, and the cap is the reason this is not a terminal state in
|
|
// code: a run that has failed twenty times may still be a Postgres that was down for an hour, and a
|
|
// mechanism that stopped looking would need a human to notice a thing that had healed. What the
|
|
// count buys is that a human is TOLD; what it must not buy is this platform deciding on its own that
|
|
// a run whose engine it could not reach is over — everywhere else in this reconciler "I could not
|
|
// ask" is never "the run is gone", and a counter in front of that mistake does not fix it. The
|
|
// terminal verdict is an operator's, through `tmplatformctl run abandon`.
|
|
func backoff(failures int) time.Duration {
|
|
const base, cap = time.Minute, 30 * time.Minute
|
|
d := base
|
|
for range min(failures, 16) - 1 {
|
|
if d >= cap {
|
|
break
|
|
}
|
|
d *= 2
|
|
}
|
|
return min(d, cap)
|
|
}
|
|
|
|
// runBudget is what ONE run may cost a pass of the sweep.
|
|
//
|
|
// It exists because the pass has a budget of its own and the work is not uniform: a spawn or a
|
|
// settlement is a `tmctl status` call, seconds of the engine's CPU each, and a handful of runs whose
|
|
// engine hangs used to spend the whole pass — after which the list, which is ordered the same way
|
|
// every time, was never reached past them. That is starvation with no upper bound, and it was
|
|
// invisible (register row PD-169; the counter that makes it visible is the sweep's own metric).
|
|
//
|
|
// Generous rather than tight: the call it bounds legitimately takes seconds on a large book, and the
|
|
// point is to bound the pathological case, not to race the ordinary one.
|
|
const defaultRunBudget = 60 * time.Second
|
|
|
|
func (s *Service) runBudget() time.Duration {
|
|
if s.Cfg.RunBudget > 0 {
|
|
return s.Cfg.RunBudget
|
|
}
|
|
return defaultRunBudget
|
|
}
|
|
|
|
// Lag is how far the furthest-behind live run's journal is beyond this platform's cursor, in bytes.
|
|
//
|
|
// It is the tailer's own health, and until this pack nothing reported it: a projection that has
|
|
// stopped moving and a run that is simply quiet look the same from outside. Read separately from the
|
|
// sweep rather than folded into it because it is telemetry — a failure to measure must not be able
|
|
// to affect what the sweep does.
|
|
func (s *Service) Lag(ctx context.Context) (int64, error) {
|
|
live, err := s.Store.ListLiveRuns(ctx)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
var worst int64
|
|
for _, l := range live {
|
|
size, err := journalSize(l.Workdir)
|
|
if err != nil {
|
|
continue // a book whose directory is unreadable is the reconciler's problem, not this one's
|
|
}
|
|
worst = max(worst, size-l.Position.Offset)
|
|
}
|
|
return worst, nil
|
|
}
|
|
|
|
// reconcile brings one live run's row into line with the world, and reports whether this pass
|
|
// ESTABLISHED anything about it — the journal moved, the engine answered, or the attempt ended.
|
|
//
|
|
// That second answer is what clears the deferral count: passes here are deliberately cheap, so "no
|
|
// error" on its own says only that nothing was asked.
|
|
func (s *Service) reconcile(ctx context.Context, l pgstore.LiveRun) (bool, error) {
|
|
// The cursor this call MOVED, not the one the sweep's snapshot was taken with. Everything below
|
|
// that asks "has the stream said anything" has to ask about now: the snapshot is read before the
|
|
// journal is drained, so a run whose first events arrive during this very sweep still looks
|
|
// silent in it — and the repair channel would then be woken for a run that had just spoken.
|
|
// Measured end to end, not reasoned about: a live run showed "edit 10/10" from a status report
|
|
// seconds after its journal said "edit 0/10".
|
|
// ⚠ A journal that cannot be read stops the PROJECTION and nothing else. It used to abort the
|
|
// whole reconcile before the exit marker was even looked at, so one malformed line left a run
|
|
// "translating" forever with its hold reserved — the engine long gone, the marker on disk, and
|
|
// every sweep failing at the same byte. The lifecycle is decided from the marker and from
|
|
// systemd; the journal only decides how fresh the numbers are.
|
|
seq, parked, drainErr := s.drainJournal(ctx, l)
|
|
// The cheapest evidence that this run is reachable, and the one a healthy run produces on nearly
|
|
// every pass.
|
|
moved := seq > l.Position.LastSeq
|
|
l.Position.LastSeq = max(l.Position.LastSeq, seq)
|
|
if drainErr != nil {
|
|
s.log().ErrorContext(ctx, "journal could not be materialized; the run's lifecycle continues without it",
|
|
"run", l.RunID, "err", drainErr)
|
|
}
|
|
marker, err := runner.ReadMarker(s.markerPath(l.RunID, l.AttemptNo))
|
|
switch {
|
|
case err == nil, errors.Is(err, runner.ErrBadMarker):
|
|
// There is a marker, so the unit ENDED, and the ending is decided here in both readings of it
|
|
// — which is why the re-read is shared rather than repeated. A ceiling the stream announced
|
|
// during this very pass outranks whatever the marker says, and putting the re-read on only the
|
|
// readable branch buried exactly that case: a ceiling halt whose marker was corrupted came out
|
|
// `failed` with no reason at all, on the one path where the exit code cannot say otherwise
|
|
// either.
|
|
status, reason, rerr := s.freshRunState(ctx, l)
|
|
if rerr != nil {
|
|
return moved, rerr // nothing is closed on facts we could not establish; the next pass tries again
|
|
}
|
|
l.PausedReason = reason
|
|
if status != "" {
|
|
// The snapshot is pre-drain; the row is not. Everything below that reasons about WHERE this
|
|
// run stands has to reason about now.
|
|
l.Status = status
|
|
}
|
|
if err != nil {
|
|
// What the marker says cannot be read. Retrying is what the register row is about: the
|
|
// write is atomic (temp+fsync+rename), so a marker that does not parse was changed by
|
|
// something outside this platform, and no number of sweeps repairs it. The run is ended
|
|
// with the fact recorded rather than re-read forever (register row PD-164).
|
|
s.log().ErrorContext(ctx, "the exit marker cannot be read; the run is closed on what is left rather than reconciled forever",
|
|
"run", l.RunID, "attempt", l.AttemptNo, "err", err)
|
|
return true, s.finish(ctx, l, runner.Marker{Unit: l.UnitName, Result: runner.UnreadableMarkerResult})
|
|
}
|
|
if interruptedBySomeoneElse(l, marker) && l.Status != "awaiting_bank" {
|
|
// Exit 5 is a caught SIGTERM, and nothing here recorded asking for one. See the function.
|
|
//
|
|
// ⚠ …unless the run is standing at the BANK-SIGNING STOP, and that exclusion is the same
|
|
// rule as the ceiling's one level down: an outside signal landing on a run that has already
|
|
// paused on purpose does not un-pause it. Without it this branch RESTARTS past the paid
|
|
// stop before `outcome` — the only other place that knows about the stop — is ever
|
|
// consulted, and after D39.158 that restart marches straight through the boundary on the
|
|
// engine's presented memory. Found by this pack's own adversarial pass, which reproduced it
|
|
// with a marker reading exit 5 on an `awaiting_bank` row.
|
|
s.log().InfoContext(ctx, "the run was signalled by something that is not this platform's stop; restarting it",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return true, s.restart(ctx, l)
|
|
}
|
|
return true, s.finish(ctx, l, marker)
|
|
case !errors.Is(err, runner.ErrNoMarker):
|
|
return moved, err
|
|
}
|
|
if l.UnitName == "" {
|
|
if l.StopRequestedAt != nil {
|
|
// Stopped before it ever started: the request was admitted, the money was held, and the
|
|
// unit was never created. There is nothing to signal and nothing to wait for, so the run
|
|
// ends here and its hold comes back whole through the ordinary settlement.
|
|
return true, s.finishStopped(ctx, l)
|
|
}
|
|
// Admitted and never spawned: the platform stopped between the transaction and the unit, or
|
|
// the queue entry was lost. The reconciler is the backstop for both.
|
|
return true, s.spawnAttempt(ctx, l)
|
|
}
|
|
alive, err := s.Runner.Alive(ctx, l.UnitName)
|
|
if err != nil {
|
|
// The bus did not answer. "I could not ask" must never be read as "the run is gone": that
|
|
// mistake restarts a live engine against its own project lock.
|
|
return moved, err
|
|
}
|
|
if alive {
|
|
if l.StopRequestedAt != nil {
|
|
// The intent is committed and the unit is still there. Either the signal never went out —
|
|
// the platform died between writing the intent and asking systemd — or the engine is
|
|
// finishing the chunk it has already paid for. Asking again is free and idempotent, and it
|
|
// is the only thing that closes the first case.
|
|
if err := s.Runner.Stop(ctx, l.UnitName); err != nil {
|
|
return moved, err
|
|
}
|
|
return true, nil
|
|
}
|
|
refreshed, err := s.maybeResync(ctx, l, parked)
|
|
return moved || refreshed, err
|
|
}
|
|
if s.now().Sub(l.AttemptStartedAt) < spawnGrace {
|
|
return moved, nil // systemd has not started it yet
|
|
}
|
|
if l.StopRequestedAt != nil {
|
|
// The unit is gone and left no marker, which is the shape of a reboot — but this run was
|
|
// asked to stop, and restarting it would spend the account's money on work its owner had
|
|
// just cancelled. The intent decides, because it is the one fact here that is ours.
|
|
return true, s.finishStopped(ctx, l)
|
|
}
|
|
status, reason, err := s.freshRunState(ctx, l)
|
|
if err != nil {
|
|
return moved, err // same rule as above: a restart is a decision, and this one is not established yet
|
|
}
|
|
if status != "" {
|
|
l.Status = status
|
|
}
|
|
if l.PausedReason = reason; l.PausedReason != "" {
|
|
// The stream already said a ceiling stopped this run, and a ceiling halt is not an
|
|
// interruption: the engine ended on purpose and the next process would meet the same limit.
|
|
// Restarting here would spend a hold on that, and it would also LOSE the reason — the restart
|
|
// ends with a pause of its own, so a run stopped by the ENGINE's daily ceiling came back
|
|
// wearing the account's. This branch is what keeps `restart` a path only interruptions take.
|
|
s.log().InfoContext(ctx, "the run's stream reported a ceiling and its unit is gone; closing it as paused",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return true, s.finish(ctx, l, runner.Marker{Unit: l.UnitName, Result: runner.UnitVanishedResult})
|
|
}
|
|
if l.Status == "awaiting_bank" {
|
|
// The stream already said this run stopped at the BANK boundary, and a signing stop is not an
|
|
// interruption for exactly the reason the ceiling above is not: the engine ended on purpose,
|
|
// at a pause the user paid for and has not answered yet. Restarting here would spend a fresh
|
|
// hold on a march straight THROUGH that pause — after D39.158 the next attempt carries
|
|
// `--verify-bank` and the engine's presented memory (storage v16) auto-continues over a map it
|
|
// has already shown, one WARN in its own journal and nothing on any surface of ours. The stop
|
|
// the user paid for would simply cease to exist, which is unified-backlog row 240.
|
|
//
|
|
// ⚠ This is what protects the paid stop now; the platform's old guard did not. That guard
|
|
// (`LiftBankStop`, removed with the workaround) only decided which BIT the restart wrote, and
|
|
// the engine of the memory era stopped consulting the flag for this question — so keying the
|
|
// protection on the argv could not hold whatever the bit said. The protection belongs where
|
|
// the decision is: whether this platform restarts the run at all.
|
|
s.log().InfoContext(ctx, "the run stands at the bank-signing stop and its unit is gone; closing it there rather than restarting past a stop nobody has answered",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return true, s.finish(ctx, l, runner.Marker{Unit: l.UnitName, Result: runner.UnitVanishedResult})
|
|
}
|
|
return true, s.restart(ctx, l)
|
|
}
|
|
|
|
// freshPausedReason re-reads what the run's projection says about a ceiling, because the sweep's
|
|
// snapshot predates the drain this pass just performed.
|
|
//
|
|
// It is asked at the two points where the answer DECIDES something — the ending of a run whose unit
|
|
// left a marker, readable or not, and the fate of one whose unit vanished without one — rather than
|
|
// after every drain: on those branches it is one query on a run that is about to change state, and
|
|
// everywhere else it would be one query per live run per tick for nothing.
|
|
//
|
|
// ⚠ A failed read is REPORTED, never absorbed, and the difference is a run's contract-visible state.
|
|
// It used to fall back to the snapshot, which is empty in exactly the case this exists for — the
|
|
// event landed in THIS drain — and the callers then closed the run on that emptiness. On the
|
|
// corrupt-marker branch there is no exit code to carry the ceiling independently and no next pass to
|
|
// correct it (the branch closes the run by design, PD-164), so one ordinary Postgres blip turned a
|
|
// resumable ceiling halt into `failed` with no reason. "I could not ask" is not an answer here any
|
|
// more than it is when systemd does not reply; the sweep logs it and the next pass reads the same
|
|
// world.
|
|
//
|
|
// ⚠ It re-reads the STATUS as well as the reason, and that is not a convenience. The bank-stop event
|
|
// writes `awaiting_bank` on the run row in the very drain this pass performs (pgstore/sink.go,
|
|
// TypeBankStop), and the snapshot is a by-value copy taken before it — so a guard that asked
|
|
// `l.Status` about the signing stop asked a value that could not yet know. Measured on the real
|
|
// journal by this pack's adversarial pass: after the drain the row read `awaiting_bank` while the
|
|
// snapshot still read `translating`, and the paid stop was restarted past anyway. The two facts are
|
|
// re-read together because they are re-read for the same reason and on the same branches.
|
|
func (s *Service) freshRunState(ctx context.Context, l pgstore.LiveRun) (status, reason string, err error) {
|
|
status, reason, err = s.Store.RunPausedReason(ctx, l.RunID)
|
|
if err != nil {
|
|
// ⚠ The wording keeps "paused reason" deliberately: `seam_test.TestAnEndingIsNeverDecidedFromAReadThatFailed`
|
|
// matches on it, and that pin is about the PROPERTY — an ending is never decided past a read
|
|
// that failed — which this function still has. Renaming the function was this pack's business;
|
|
// silently retiring somebody else's assertion is not.
|
|
return "", "", fmt.Errorf("runs: the run's status and paused reason could not be re-read after the drain: %w", err)
|
|
}
|
|
return status, reason, nil
|
|
}
|
|
|
|
// finishStopped closes a run whose end this platform asked for and whose unit left no marker to read:
|
|
// it was never created, or the manager went away with it.
|
|
//
|
|
// The two cases are not the same write. Where a unit EXISTED the ordinary finish applies. Where it
|
|
// never did, the close has to re-check that under a lock: the queue worker may have claimed the
|
|
// attempt between this sweep's snapshot and now, and closing the run then strands a live engine that
|
|
// no list looks at any more.
|
|
func (s *Service) finishStopped(ctx context.Context, l pgstore.LiveRun) error {
|
|
if l.StopRequestedAt == nil {
|
|
// The SNAPSHOT may predate the request — that is the whole reason one of the two callers is
|
|
// here: the reconciler read the run, spent seconds settling, and the stop landed inside that
|
|
// window. Reaching this function is itself the establishment of the fact, so the local copy is
|
|
// made to say so rather than letting `outcome` read a nil and call the ending a failure.
|
|
now := s.now()
|
|
l.StopRequestedAt = &now
|
|
}
|
|
if l.UnitName != "" {
|
|
return s.finish(ctx, l, runner.Marker{Unit: l.UnitName, Result: pgstore.StopRequestedResult})
|
|
}
|
|
// ⚠ An empty unit name is not proof that no process exists. `ReleaseSpawnClaim` clears the name
|
|
// when the unit could NOT be created — and "could not be created" is not "was not created": a
|
|
// systemd-run killed after it had already asked leaves an engine running, which is why the spend
|
|
// baseline is deliberately kept on such an attempt (RecordSpawn). That baseline is the tombstone,
|
|
// and where it exists the unit's name is derivable, so systemd is asked rather than assumed.
|
|
// Closing over a live engine would leave it spending against a run this platform calls finished.
|
|
if l.SpendBaseline != nil {
|
|
unit := unitName(l.RunID, l.AttemptNo)
|
|
alive, err := s.Runner.Alive(ctx, unit)
|
|
if err != nil {
|
|
// "I could not ask" is never "the run is gone" — the same rule the reconciler follows
|
|
// everywhere else.
|
|
return err
|
|
}
|
|
if alive {
|
|
s.log().InfoContext(ctx, "a claim was given back but its unit exists; stopping it instead of closing the run",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return s.Runner.Stop(ctx, unit)
|
|
}
|
|
}
|
|
finished, err := s.Store.FinishUnspawnedStop(ctx, l.RunID, l.AttemptID, s.now())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if !finished {
|
|
// It was spawned after all. Nothing to do: the next pass sees a live unit and an intent that
|
|
// says to stop it.
|
|
s.log().InfoContext(ctx, "the stopped run was spawned after this sweep read it", "run", l.RunID)
|
|
return nil
|
|
}
|
|
s.log().InfoContext(ctx, "run finished", "run", l.RunID, "status", "stopped",
|
|
"result", pgstore.StopRequestedResult)
|
|
_, err = s.settle(ctx, l)
|
|
return err
|
|
}
|
|
|
|
// drainJournal applies whatever the engine has written since the cursor, and returns where the
|
|
// cursor now stands.
|
|
func (s *Service) drainJournal(ctx context.Context, l pgstore.LiveRun) (seq int64, parked bool, err error) {
|
|
if l.Quarantined {
|
|
return l.Position.LastSeq, false, nil
|
|
}
|
|
path := filepath.Join(l.Workdir, ingest.JournalFile)
|
|
sink := s.Store.NewRunSink(l.AttemptID, l.RunID, l.BookID)
|
|
from := ingest.Position{Offset: l.Position.Offset, LastSeq: l.Position.LastSeq, LastHash: l.Position.LastHash}
|
|
pos, _, err := ingest.Tail(ctx, path, l.EngineRunID, from, sink)
|
|
switch {
|
|
case errors.Is(err, ingest.ErrForeignStreamAhead):
|
|
// The stream of this attempt has ended in this journal and a stranger's begins. NOT a
|
|
// quarantine — nothing is corrupt and nothing was misread — but not "caught up" either: the
|
|
// cursor stays on that handshake, so the projection has nothing more to say from here, and
|
|
// the run's freshness has to come from the repair channel instead. Said out loud because the
|
|
// state carries no other signal: the attempt is not quarantined, so the gauge does not count
|
|
// it and the operator's listing shows an empty QUARANTINE cell.
|
|
//
|
|
// ⚠ It is NOT evidence that our process is gone. A respawn of this same attempt is handed the
|
|
// same stream id by the platform (runs.engineStreamID keys on run and attempt, not on the
|
|
// try) and the engine mints a fresh one when that id has already written for this book, so
|
|
// the "stranger" may be this very run, alive and writing.
|
|
if s.sayParked(l.AttemptID) {
|
|
s.log().WarnContext(ctx, "the journal continues under another stream id; this attempt's projection stops here and its freshness falls back to the repair channel",
|
|
"run", l.RunID, "attempt", l.AttemptNo, "offset", pos.Offset)
|
|
}
|
|
return pos.LastSeq, true, nil
|
|
case errors.Is(err, ingest.ErrNoJournal):
|
|
// The engine writes its first line whenever it gets there, and a run whose unit systemd has
|
|
// not started yet has no journal at all. Ordinary, and not a failure: the tailer waits.
|
|
return l.Position.LastSeq, false, nil
|
|
case err != nil && !quarantines(err):
|
|
// A moment we could not read it, not a stream we cannot read. Returned so the sweep logs it and
|
|
// meets the same bytes again next pass.
|
|
return l.Position.LastSeq, false, err
|
|
case err != nil:
|
|
// The RUN is left alone: it is spending money the account reserved, and our inability to read
|
|
// its journal is not a reason to throw that away. Freshness falls back to the resync channel.
|
|
s.log().ErrorContext(ctx, "journal cannot be materialized; falling back to resync",
|
|
"run", l.RunID, "err", err)
|
|
return l.Position.LastSeq, false, s.Store.Quarantine(ctx, l.AttemptID, err.Error())
|
|
}
|
|
if pos.Offset > l.Position.Offset {
|
|
// Lines that were read and NOT applied — duplicates the cursor already covers — still move
|
|
// the byte hint. Without this they are re-read on every sweep, forever.
|
|
//
|
|
// ⚠ There is one way for that to be a symptom rather than housekeeping, and it is free to
|
|
// notice here: lines went past, none of them were ours, and none ever has been. The platform
|
|
// names the stream before the unit starts and hands the name over as TM_TRACE_ID, so an
|
|
// engine build that does not accept it announces itself under an id this cursor will never
|
|
// match — and the run would then go quiet for its whole life, falling back to the resync
|
|
// channel with nothing saying why.
|
|
if l.EngineRunID != "" && pos.LastSeq == 0 {
|
|
s.log().WarnContext(ctx, "the journal has lines and none belong to this attempt's stream; the engine may not be honouring the run id it was given (falling back to resync)",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
}
|
|
return pos.LastSeq, false, s.Store.SaveCursor(ctx, l.AttemptID, pgstore.Position{Offset: pos.Offset})
|
|
}
|
|
return pos.LastSeq, false, nil
|
|
}
|
|
|
|
// quarantines decides what a failure to materialize the journal MEANS: a stream this platform cannot
|
|
// read, or a moment in which it could not read one.
|
|
//
|
|
// That distinction is the whole decision, and getting it wrong is expensive in both directions. A
|
|
// gap, a changed payload, a malformed line, a line past the buffer: none is repaired by reading the
|
|
// same bytes again, and retrying them forever is what wedged a run — so those stop the projection.
|
|
// Our own shutdown and a lock Postgres broke are the opposite: the very next sweep reads exactly the
|
|
// same bytes successfully, and stopping the projection over one blinds a live, paying run for good.
|
|
func quarantines(err error) bool {
|
|
switch {
|
|
case err == nil:
|
|
return false
|
|
case errors.Is(err, context.Canceled), errors.Is(err, context.DeadlineExceeded):
|
|
return false
|
|
case pgstore.IsTransient(err):
|
|
return false
|
|
default:
|
|
return true
|
|
}
|
|
}
|
|
|
|
// maybeResync refreshes a live run from `tmctl status --json`, at most once every ResyncEvery.
|
|
//
|
|
// ⚠ What it produces is HONESTLY STALE, and by an amount the interval names: the run moves
|
|
// continuously and this reads it every few minutes. Now that the emitter writes (D39.131) it is what
|
|
// it was designed to be — the REPAIR path, reached only where the stream has said nothing or has
|
|
// been quarantined — rather than the only source of progress it had to be until then.
|
|
//
|
|
// It reports whether the engine actually ANSWERED: on the passes it skips — nearly all of them —
|
|
// nothing has been established, and the caller must not read that silence as health.
|
|
func (s *Service) maybeResync(ctx context.Context, l pgstore.LiveRun, parked bool) (bool, error) {
|
|
// The stream, when there is one, is the FRESHER source and the free one. A status call costs
|
|
// seconds of CPU on the engine's side, every time, and can only report what the journal has
|
|
// already said — so the repair channel runs where there is nothing to repair from: an attempt
|
|
// whose cursor has never moved, or one whose materialization was quarantined and for which this
|
|
// is now the only source.
|
|
//
|
|
// ⚠ What a resync DOES materialise is narrower than this comment used to claim, and the claim is
|
|
// corrected rather than inherited: "ApplyStatus now materializes the same four counters as the
|
|
// stream" was true of four columns nothing ever read (PD-411, dropped in migration 00031). The
|
|
// bar is derived from `chapters`, which this channel does not write — so a run whose stream is
|
|
// quarantined keeps the progress its stream last delivered, however faithfully the engine answers
|
|
// here. The repair channel repairs the ETA, the freshness stamp and the wave shape. That gap has
|
|
// a register row of its own.
|
|
// `parked` stands beside `Quarantined` and for the same reason: in both states the stream has
|
|
// stopped speaking for this attempt, so the run's numbers can only come from here. The difference
|
|
// is that a quarantine is written down and a park is not — it lives for the length of one pass —
|
|
// which is why the caller carries it rather than the row.
|
|
if l.Position.LastSeq > 0 && !l.Quarantined && !parked {
|
|
return false, nil
|
|
}
|
|
if !s.dueForResync(l.RunID) {
|
|
return false, nil
|
|
}
|
|
now := s.now()
|
|
rep, err := s.Engine.Status(ctx, s.engineBinary(l), l.Workdir)
|
|
if err != nil {
|
|
// A status call that fails is not a run that failed. It is logged and retried on the next
|
|
// interval; the read model keeps the last figures it had.
|
|
//
|
|
// And not evidence: this is the call that hangs on a wedged project, so answering "established"
|
|
// would clear the count of the run that just spent the pass.
|
|
s.log().WarnContext(ctx, "resync failed", "run", l.RunID, "err", err)
|
|
return false, nil
|
|
}
|
|
return true, s.Store.ApplyStatus(ctx, l.RunID, l.BookID, rep, now)
|
|
}
|
|
|
|
// dueForResync is the rate limit, and it is the whole of what makes the resync affordable: the
|
|
// sweep runs every few seconds and each status call costs seconds of CPU. It records the decision as
|
|
// it makes it, so two sweeps cannot both conclude "due".
|
|
//
|
|
// In memory on purpose: this is a rate limit, not a fact — losing it on restart costs one extra
|
|
// status call, whereas persisting it would put a row write on a path that exists to avoid work.
|
|
func (s *Service) dueForResync(runID string) bool {
|
|
if s.Cfg.ResyncEvery <= 0 || s.Engine == nil {
|
|
return false
|
|
}
|
|
now := s.now()
|
|
if s.resynced == nil {
|
|
s.resynced = map[string]time.Time{}
|
|
}
|
|
if last, ok := s.resynced[runID]; ok && now.Sub(last) < s.Cfg.ResyncEvery {
|
|
return false
|
|
}
|
|
s.resynced[runID] = now
|
|
return true
|
|
}
|
|
|
|
// sayParked rate-limits the parked attempt's line: true at the crossing, then no more often than the
|
|
// repair channel itself speaks.
|
|
//
|
|
// Neither extreme is right for this state, which is why it is throttled rather than silenced or left
|
|
// alone. A park can last the run's whole life and it is written down NOWHERE — no column, no gauge,
|
|
// no listing — so a once-only line leaves an operator who starts watching afterwards with nothing;
|
|
// while a line every fifteen seconds is the one they filter, which this file already says of the
|
|
// deferral (deferItem). The interval is the resync's on purpose: it is the rate at which anything
|
|
// about a parked run changes at all.
|
|
func (s *Service) sayParked(attemptID int64) bool {
|
|
every := s.Cfg.ResyncEvery
|
|
if every <= 0 {
|
|
every = time.Minute
|
|
}
|
|
now := s.now()
|
|
if s.parkedSaid == nil {
|
|
s.parkedSaid = map[int64]time.Time{}
|
|
}
|
|
if last, ok := s.parkedSaid[attemptID]; ok && now.Sub(last) < every {
|
|
return false
|
|
}
|
|
s.parkedSaid[attemptID] = now
|
|
return true
|
|
}
|
|
|
|
// finish closes a run whose unit has ended.
|
|
func (s *Service) finish(ctx context.Context, l pgstore.LiveRun, m runner.Marker) error {
|
|
delete(s.resynced, l.RunID) // a finished run keeps no rate-limit entry: the map is per process
|
|
delete(s.parkedSaid, l.AttemptID)
|
|
status, pausedReason, exit := outcome(l, m)
|
|
closed, err := s.Store.FinishRun(ctx, pgstore.RunEnding{
|
|
RunID: l.RunID,
|
|
AttemptID: l.AttemptID,
|
|
Status: status,
|
|
ExitResult: m.Result,
|
|
ExitCode: exit,
|
|
// Written with the status and not by a second call: a `paused` run whose reason lands one
|
|
// statement later is a run the resume path can read in between and misjudge. The failure
|
|
// reason is written the same way and for the same reason — it is the one state that IS an
|
|
// error, and a client decides a retry from it.
|
|
PausedReason: pausedReason,
|
|
FailureReason: failureReason(status, exit, m),
|
|
Now: s.now(),
|
|
})
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if !closed {
|
|
// The run was already finished, or this pass is holding a snapshot of an attempt that is no
|
|
// longer the live one — a stop and a resume can both have happened since it was taken. Doing
|
|
// anything further here would settle the money of a run somebody else is now running.
|
|
s.log().InfoContext(ctx, "the run moved on since this pass read it; nothing to finish",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return nil
|
|
}
|
|
s.log().InfoContext(ctx, "run finished", "run", l.RunID, "status", status, "result", m.Result)
|
|
if l.Resnapshot && status == "ready" {
|
|
// The bank-move fact is retired EXPLICITLY, and only here: a run that carried the consents
|
|
// and finished clean has demonstrably walked the correction into the text. Every other
|
|
// ending — failed, stopped, paused — leaves the fact standing, so the next admission still
|
|
// carries the flags instead of dying on the engine's snapshot guard (adversarial K6: a
|
|
// timestamp predicate let a failed run retire the fact and loop the guard forever). The
|
|
// write is outside FinishRun's transaction on purpose: if it fails, the stale fact costs
|
|
// one harmless --resnapshot on the next run, never a dead one.
|
|
if err := s.Store.ClearBankMove(ctx, l.BookID); err != nil {
|
|
s.log().ErrorContext(ctx, "the finished re-snapshot run could not retire the bank-move fact", "err", err)
|
|
}
|
|
}
|
|
// Settling immediately rather than waiting for the next sweep: the hold is the account's money
|
|
// and every second it stays reserved is a second the user cannot start another book. The text the
|
|
// run produced is NOT materialized here: `FinishRun` recorded that debt in the transaction that
|
|
// closed the run, and the materializer's own pass pays it — two full re-chunks of the source,
|
|
// inside this one, held up every account's settlement behind it.
|
|
//
|
|
// The verdict is discarded HERE and only here: this settlement is opportunistic — the attempt is
|
|
// on the reconciliation phase's list, not the settlement phase's — and the settlement phase is
|
|
// what counts a failure and defers. Counting it twice for one pass would halve the threshold.
|
|
_, err = s.settle(ctx, l)
|
|
return err
|
|
}
|
|
|
|
// outcome maps what systemd saw onto the product status of the run, and onto the reason it carries.
|
|
//
|
|
// The engine's exit contract is the input for a process that exited on its own (ingest/exit.go); a
|
|
// process that did NOT exit on its own has no exit code to map, and $SERVICE_RESULT is what tells a
|
|
// stop we asked for from a kill we did not.
|
|
//
|
|
// ⚠ The CEILING branch is the whole of PD-113, and it reads two independent channels because they
|
|
// fail independently: the stream's `ceiling` event and exit code 4. A journal that could not be
|
|
// written still leaves the code; a process killed before it could exit still leaves the event. Both
|
|
// answer `paused`, and neither may answer `failed` — the stop is resumable, and the contract forbids
|
|
// it (§BookStatus). Before the emitter landed there was only the event, and a ceiling halt with no
|
|
// journal was recorded as an infrastructure failure.
|
|
//
|
|
// A stop this platform ASKED for is a different question and is answered here, because the answer is
|
|
// ours: exit 5 says a signal wound the run down and says nothing about WHO sent it, so the record
|
|
// written before the signal went out is what tells a user's stop from a reboot (register row
|
|
// PD-152). What that record's ABSENCE means is decided one level up — see interruptedBySomeoneElse.
|
|
func outcome(l pgstore.LiveRun, m runner.Marker) (status, pausedReason string, exitCode *int) {
|
|
code, exited := m.Exited()
|
|
if exited {
|
|
switch ingest.OutcomeOf(code) {
|
|
case ingest.OutcomeClean, ingest.OutcomeFlagged:
|
|
// Flagged is a finished translation whose notes carry the flags. The engine's own clean
|
|
// answer wins over a stop that arrived while it was already done.
|
|
return "ready", "", &code
|
|
case ingest.OutcomeBankStop:
|
|
// The bank-signing stop. The ATTEMPT is over — the engine exits — and the run is left in a
|
|
// state `resume` continues from, with the decisions AS THEY STAND: signing is one act over
|
|
// the whole bank and the completeness gate that used to stand here was removed with D39.144.
|
|
return "awaiting_bank", "", &code
|
|
case ingest.OutcomeStopped:
|
|
// A caught signal over a run this platform asked to stop. WHICH of the two records is
|
|
// older decides nothing here, and reading the clock instead recorded `failed` for a stop
|
|
// the user did ask for: the guard below exists so that a run which
|
|
// finished BY ITSELF a moment before the click is not called cancelled — and a run that
|
|
// finished by itself does not exit 5. Its own endings are answered above, in this same
|
|
// switch, before anything looks at a timestamp. What is left under exit 5 with an intent
|
|
// on file is a stop, whether the signal that wound it down was ours or a reboot's that
|
|
// arrived first; the two clocks being compared are not even the same one — the marker
|
|
// carries the host's, the intent the platform's.
|
|
if l.StopRequestedAt != nil {
|
|
return "stopped", "", &code
|
|
}
|
|
}
|
|
}
|
|
// ⚠ THE USER'S OWN STOP WINS OVER A CEILING that arrived in the same drain (register row PD-241,
|
|
// ratified D39.132 п.2д as "the next touch of this zone", which is this pack).
|
|
//
|
|
// The reason it is not cosmetic: a ceiling reason of `credit_exhausted` lights the ACCOUNT-level
|
|
// halted flag (`ReadUsage` keys on exactly that value), so a user who pressed stop on an account
|
|
// with 97% of its balance left was told their credit had run out. Reproduced by review.
|
|
//
|
|
// It is deliberately placed AFTER the engine's own endings above and BEFORE the ceiling below:
|
|
// a run that finished BY ITSELF a moment before the click is not cancelled — those exits are
|
|
// answered in the switch — while everything that is genuinely an interruption belongs to whoever
|
|
// asked for it, and we are the only side that records having asked.
|
|
if stoppedOnRequest(l, m) {
|
|
return "stopped", "", exitCodeOrNil(code, exited)
|
|
}
|
|
if l.PausedReason != "" {
|
|
// The stream said this run stopped on a ceiling, and that survives ANY exit: the process may
|
|
// have been killed on its way out, and the fact does not become less true.
|
|
return "paused", l.PausedReason, nil
|
|
}
|
|
if l.Status == "awaiting_bank" {
|
|
// Same rule, same reason, for the bank-signing stop: the journal's bank-stop event moved this
|
|
// run to the pause BEFORE any marker was read, and a process killed on its way out of that
|
|
// pause does not un-pause it. Without this the run would be recorded `failed`/`interrupted`
|
|
// and the paid stop would be answered with a retry — the very march past the boundary that
|
|
// unified-backlog row 240 names. It sits AFTER the engine's own endings (a clean exit above
|
|
// wins: a run that finished by itself is not waiting for a signature) and after the user's
|
|
// own stop, whose word is ours and outranks a pause we are only reporting.
|
|
return "awaiting_bank", "", exitCodeOrNil(code, exited)
|
|
}
|
|
if exited && ingest.OutcomeOf(code) == ingest.OutcomeCeiling {
|
|
// No event reached us — the branch above would have taken it — so WHICH ceiling stopped the
|
|
// run is genuinely unknown here, and that is what gets recorded. It is not a rare corner: a
|
|
// quarantined projection never drains at all, so every ceiling halt of such a run arrives
|
|
// this way. Guessing `credit_exhausted` would light the account-level halted flag for an
|
|
// account that may have plenty of money (register row PD-203), and guessing `daily_ceiling`
|
|
// would refuse a resume the user could have completed.
|
|
return "paused", pgstore.PausedCeilingUnknown, &code
|
|
}
|
|
if !exited && m.Result == "success" {
|
|
// Killed by a signal and systemd calls the result success: a stop nobody recorded, and no exit
|
|
// code at all to read it from.
|
|
return "stopped", "", nil
|
|
}
|
|
if exited {
|
|
// Everything left, including the REFUSAL band. A refusal did no work and spent nothing, and
|
|
// `failed` is the honest product status for it: the run cannot proceed, and the account is
|
|
// charged the nothing it cost. It is deliberately NOT restarted — a refusal is reproducible by
|
|
// construction (a configuration that will not load, a project another process holds), so a
|
|
// restart would re-take a hold and meet the same answer on every sweep, forever. The class is
|
|
// in exit_code for an operator, and the intake — the one path that acts destructively on a
|
|
// refusal — reads the band itself (books.intakeReason).
|
|
return "failed", "", &code
|
|
}
|
|
return "failed", "", nil
|
|
}
|
|
|
|
// failureReason is why a run ended in `failed`, in the contract's own three-valued vocabulary
|
|
// (canon §RunFailureReason). Empty for every other status: a run that is not failed carrying a
|
|
// reason it failed would be read by a client as one that did.
|
|
//
|
|
// The three answer ONE question — is a retry worth offering — which is why they are coarse:
|
|
//
|
|
// - `source_unreadable`: the engine read the book and found nothing to translate in it (exit 11,
|
|
// the one refusal that is about the user's text). Adding the file again in another form is the
|
|
// remedy; retrying this run is not.
|
|
// - `service_error`: everything else the engine ANSWERED with — a configuration that will not
|
|
// load, a project another process holds, a schema an upgrade has not migrated, a refusal class
|
|
// this build has never heard of, and a plain exit 1. All of them are about the deployment, and
|
|
// retrying alone does not clear any of them.
|
|
// - `interrupted`: the run ended without saying how — killed by a signal nobody recorded, or a
|
|
// marker with no exit code at all. Retrying IS the remedy, and finished work is not bought
|
|
// again.
|
|
func failureReason(status string, exit *int, m runner.Marker) string {
|
|
if status != "failed" {
|
|
return ""
|
|
}
|
|
if exit != nil {
|
|
if *exit == ingest.ExitSourceUnreadable {
|
|
return "source_unreadable"
|
|
}
|
|
return "service_error"
|
|
}
|
|
// ⚠ No exit code at all, and systemd's own word for what happened is what separates the two
|
|
// answers. It was IGNORED here — the marker was taken as a parameter and never read — so an
|
|
// out-of-memory kill and a stop-timeout SIGKILL both told the client that retrying would help,
|
|
// which for those two is exactly false: the next spawn dies the same way. `interrupted` is left for what it means — an ending nobody described.
|
|
switch m.Result {
|
|
case "oom-kill", "timeout", "core-dump", "watchdog", "resources", "protocol", "exec-condition":
|
|
return "service_error"
|
|
}
|
|
return "interrupted"
|
|
}
|
|
|
|
// exitCodeOrNil is the engine's own code where it exited on its own and nothing where it did not: a
|
|
// run killed by a signal has no code to record, and recording a zero would read as a clean exit.
|
|
func exitCodeOrNil(code int, exited bool) *int {
|
|
if !exited {
|
|
return nil
|
|
}
|
|
return &code
|
|
}
|
|
|
|
// interruptedBySomeoneElse reports an attempt that a signal wound down without this platform having
|
|
// asked for one.
|
|
//
|
|
// Exit 5 means the engine caught SIGINT/SIGTERM and shut down cleanly. It does NOT say who sent it,
|
|
// and there are two senders that leave no record: an ordinary host reboot, where the user manager
|
|
// stops every unit it owns, and an operator running `systemctl --user stop` by hand. The marker
|
|
// cannot tell them apart — both are `exit-code/exited/5` — so this is a decision and not a
|
|
// deduction, and it is written down as one.
|
|
//
|
|
// It reads as an INTERRUPTION: not a failure, not the user's stop, and therefore a candidate for the
|
|
// restart path that unified backlog row 138 exists for. The argument is asymmetry. Reading a reboot
|
|
// as a user's stop leaves every run on the host dead after a routine restart, with its budget
|
|
// unspent and nobody to resume it — which is the remaining half of PD-152 and the one thing row 138
|
|
// was built to prevent. Reading a hand-stop as an interruption costs one restart of a run whose
|
|
// owner never asked for it to end, and the documented way to end a run — the contract's stop handle
|
|
// — records an intent that this function then honours. The engine's own money is untouched either
|
|
// way: the restart settles the old attempt at what it actually spent and gives the new one what is
|
|
// left of the run's budget.
|
|
//
|
|
// A ceiling halt is never this: it has its own code and its own event, and both are read before.
|
|
func interruptedBySomeoneElse(l pgstore.LiveRun, m runner.Marker) bool {
|
|
if l.StopRequestedAt != nil || l.PausedReason != "" {
|
|
return false
|
|
}
|
|
code, exited := m.Exited()
|
|
return exited && code == ingest.ExitStopped
|
|
}
|
|
|
|
// stoppedOnRequest reports whether this platform's own stop is what ended the attempt.
|
|
//
|
|
// The timestamp comparison is the race guard the design owes: a run that finished by itself a moment
|
|
// before someone pressed stop was not stopped by them, and calling it `stopped` would hide that
|
|
// ending behind a cancelled one. A marker with no timestamp is evidence we do not have, and then the
|
|
// intent — the only fact left — decides.
|
|
//
|
|
// ⚠ What reaches it is narrower than it looks, and deliberately so: every ending the engine has a
|
|
// code for is answered in `outcome`'s switch first, including the caught signal itself. So the
|
|
// clocks are compared only where the engine said something this platform cannot classify — an exit
|
|
// 1, a refusal, a process that did not exit at all — and where the comparison being wrong costs a
|
|
// label rather than a lifecycle decision.
|
|
func stoppedOnRequest(l pgstore.LiveRun, m runner.Marker) bool {
|
|
if l.StopRequestedAt == nil {
|
|
return false
|
|
}
|
|
return m.At.IsZero() || !l.StopRequestedAt.After(m.At)
|
|
}
|
|
|
|
// settle resolves the money of an attempt whose process is gone.
|
|
//
|
|
// The figure is the engine's OWN committed spend, read through `tmctl status --json` — the ratified
|
|
// repair channel — after the process has exited, so there is no lock to contend with and no
|
|
// inference from a stream that may have been truncated. If it cannot be read the reservation stays
|
|
// OPEN and the next sweep tries again: a hold that is still reserved is visible and recoverable,
|
|
// whereas a settlement against a guessed number is neither.
|
|
//
|
|
// ⚠ What this deliberately does NOT do is the escrow half of research/25 — write-ahead intent, the
|
|
// `uncertain` state, `closing` until the terminal seq. That is unified backlog row 136 and its own
|
|
// prompt; building half of it here would put a second, weaker answer next to the one being designed.
|
|
func (s *Service) settle(ctx context.Context, l pgstore.LiveRun) (settleVerdict, error) {
|
|
if s.Engine == nil {
|
|
// No engine to ask and nothing this deployment can ever settle: a read replica. Not blocked —
|
|
// there is no money path here to be stuck in.
|
|
return settlementClosed, nil
|
|
}
|
|
key := pgstore.ReservationKey(l.RunID, l.AttemptNo)
|
|
// An attempt with no unit never reached the engine, so there is nothing to ask it about — and
|
|
// asking anyway is what kept such a hold reserved for good on the very hosts that produce this
|
|
// case: a deployment whose engine cannot be run fails `Status` on every pass, and the money of a
|
|
// run that never ran would wait for an answer that never comes. The row is re-checked under the
|
|
// lock inside ReleaseUnspawned, so a snapshot that was spawned in the meantime is refused there
|
|
// and settles the ordinary way below.
|
|
if l.UnitName == "" && l.SpendBaseline == nil {
|
|
switch err := s.Store.ReleaseUnspawned(ctx, key, l.AttemptID, s.now()); {
|
|
case err == nil, errors.Is(err, pgstore.ErrNoReservation):
|
|
return settlementClosed, s.Store.MarkSettled(ctx, l.RunID, s.now())
|
|
case !errors.Is(err, pgstore.ErrAttemptSpawned):
|
|
return settlementBlocked, err
|
|
}
|
|
s.log().InfoContext(ctx, "settlement deferred: the attempt was spawned after this sweep read it",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return settlementRaced, nil // the next sweep sees the unit and settles against its baseline
|
|
}
|
|
rep, err := s.Engine.Status(ctx, s.engineBinary(l), l.Workdir)
|
|
if err != nil {
|
|
// BLOCKED and not merely deferred: the commonest permanent shape of this is a pinned engine
|
|
// binary that a rollout removed, and asking it again every fifteen seconds forever told
|
|
// nobody anything (PD-384).
|
|
s.log().WarnContext(ctx, "settlement deferred: the engine's committed spend could not be read",
|
|
"run", l.RunID, "err", err)
|
|
return settlementBlocked, nil
|
|
}
|
|
if rep.Spend == nil {
|
|
// Absent is not zero (PD-40): a settlement computed from a missing figure would release the
|
|
// whole hold and charge nothing.
|
|
s.log().WarnContext(ctx, "settlement deferred: the status report carries no committed spend",
|
|
"run", l.RunID)
|
|
return settlementBlocked, nil
|
|
}
|
|
// What the book's counter says NOW, bounded by what it said when a later attempt of the same book
|
|
// started. Settlement is allowed to defer and a deferred one can be overtaken: the run is finished,
|
|
// so nothing stops the account from starting another on the same book, and the counter this reads
|
|
// is the BOOK's lifetime total. Without the bound the deferred settlement of the first run pays for
|
|
// the second one's work as well, and the second then pays for it again.
|
|
bound, err := s.Store.SpendBound(ctx, l.BookID, l.AttemptID)
|
|
if err != nil {
|
|
return settlementBlocked, err
|
|
}
|
|
committed := *rep.Spend
|
|
if bound != nil && *bound < committed {
|
|
committed = *bound
|
|
}
|
|
if l.SpendBaseline != nil && committed < *l.SpendBaseline {
|
|
// The book's meter went BACKWARDS across this attempt, which no run produces: it is a project
|
|
// database that was replaced or restored. attemptSpend clamps to zero rather than paying an
|
|
// account for it, and this says so out loud — a settlement of nothing is not something to
|
|
// discover from a balance. WARN and not INFO because the subject is money; the figures
|
|
// themselves stay out of the line (D39.84).
|
|
s.log().WarnContext(ctx, "the book's meter reads below this attempt's own baseline; it is settled at nothing",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
}
|
|
spent, err := attemptSpend(l, committed)
|
|
if err != nil {
|
|
if l.UnitName == "" {
|
|
// No baseline because there was no spawn: the attempt was admitted and never started, so it
|
|
// cost nothing and its hold comes back WHOLE. Without this the money of a run that never ran
|
|
// is reserved forever.
|
|
//
|
|
// "Never started" is re-checked against the ROW, not against this snapshot: the sweep read
|
|
// its list before working through it, and a run can be spawned, spend and exit inside that
|
|
// window. Measured: the hold of an attempt that spent $0.50 came back in full.
|
|
switch err := s.Store.ReleaseUnspawned(ctx, key, l.AttemptID, s.now()); {
|
|
case errors.Is(err, pgstore.ErrAttemptSpawned):
|
|
s.log().InfoContext(ctx, "settlement deferred: the attempt was spawned after this sweep read it",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return settlementRaced, nil // the next sweep sees the unit and settles against its baseline
|
|
case err != nil && !errors.Is(err, pgstore.ErrNoReservation):
|
|
return settlementBlocked, err
|
|
}
|
|
return settlementClosed, s.Store.MarkSettled(ctx, l.RunID, s.now())
|
|
}
|
|
// Spawned, but with no baseline — an attempt from before this column existed. Charging the
|
|
// book's lifetime total would bill every earlier run again and charging zero would give this
|
|
// one away, so it is left open and said out loud.
|
|
// PERMANENT by construction — the column will not appear on a row already written — so this is
|
|
// the case that most needed a counter in front of it: it can never resolve on its own, and
|
|
// before PD-384 it never reached the threshold that tells an operator so. The other permanent
|
|
// shape is a pinned engine build removed at a rollout; a project REPLACED under the platform
|
|
// is NOT one of them, though an earlier draft of the handle's doc listed it — that case is
|
|
// caught above by the meter reading below its own baseline, and settled at nothing.
|
|
s.log().ErrorContext(ctx, "settlement withheld: this attempt ran without a spend baseline",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return settlementBlocked, nil
|
|
}
|
|
if err := s.Store.Settle(ctx, key, spent, s.now()); err != nil {
|
|
if errors.Is(err, pgstore.ErrNoReservation) {
|
|
// Already closed by an earlier pass. Settling is idempotent by refusal, not by repetition.
|
|
return settlementClosed, s.Store.MarkSettled(ctx, l.RunID, s.now())
|
|
}
|
|
return settlementBlocked, err
|
|
}
|
|
return settlementClosed, s.Store.MarkSettled(ctx, l.RunID, s.now())
|
|
}
|
|
|
|
// restart brings back a run whose unit vanished without a word — the case a reboot produces, since
|
|
// transient units do not survive one (research/25 §Форма, unified backlog row 138).
|
|
//
|
|
// The money is closed on the old attempt and reopened on the new one rather than carried over: the
|
|
// old hold was taken for a process that is gone, and what the account still owes for it is the
|
|
// engine's committed figure, not the ceiling. What remains of the run's budget is what the new
|
|
// attempt gets, so a restart cannot spend the run's ceiling twice.
|
|
//
|
|
// The LIMIT the resumed process is given is not that remainder: spawnAttempt reads the engine's
|
|
// meter again and hands it committed + the remainder, because the flag caps the book's cumulative
|
|
// spend and the interrupted attempt moved it (D39.122, and meter.bookCap for why the reserved figure
|
|
// is deliberately not in that sum).
|
|
func (s *Service) restart(ctx context.Context, l pgstore.LiveRun) error {
|
|
switch v, err := s.settle(ctx, l); {
|
|
case err != nil:
|
|
return err
|
|
case v == settlementBlocked:
|
|
// ⚠ Returned HERE rather than fallen through, and the early return is not tidiness (PD-424).
|
|
// `reopen`'s first act is to ask whether the previous attempt's reservation is still open, and
|
|
// in this state the answer is guaranteed to be yes — so the old path paid for a query to learn
|
|
// what `settle` had just said, then logged a WARN blaming the RESTART for a blockage that
|
|
// belongs to the settlement, and finally answered `nil`, which the caller counts as a
|
|
// successful pass.
|
|
//
|
|
// A shutdown is not a failure: the same guard the settlement phase carries (settleOne).
|
|
// Stopping the daemon cancels every item in flight, and a count a SHUTDOWN invented is the
|
|
// exact defect that phase found and named.
|
|
if ctx.Err() == nil {
|
|
return errSettlementBlocked
|
|
}
|
|
return nil
|
|
}
|
|
next, v, err := s.reopen(ctx, l, fromALiveRun)
|
|
if errors.Is(err, pgstore.ErrStopRequested) {
|
|
// The user pressed stop while this pass was settling. The intent outranks the restart, and it
|
|
// is still there — the refusal is what kept it — so the run ends here instead.
|
|
s.log().InfoContext(ctx, "restart abandoned: the run was asked to stop", "run", l.RunID)
|
|
return s.finishStopped(ctx, l)
|
|
}
|
|
if errors.Is(err, pgstore.ErrRunFinished) {
|
|
// Another generation of the sweep closed the run while this one was settling. Its snapshot is
|
|
// simply old, and there is nothing left to restart — bringing it back would take a second hold
|
|
// and start an engine for a run whose owner has been told it ended.
|
|
s.log().InfoContext(ctx, "restart abandoned: the run was finished by another pass", "run", l.RunID)
|
|
return nil
|
|
}
|
|
if err != nil {
|
|
return err
|
|
}
|
|
switch v {
|
|
case ceilingSpent, creditUnavailable:
|
|
// ⛔ TWO VERDICTS, TWO WORDS, and until 05.09 they shared one — which is PD-446 at this site.
|
|
// `ceilingSpent` is the run having spent the ORDER it was sold; `creditUnavailable` is the
|
|
// ACCOUNT being unable to carry the rest of it. Both leave the run `paused` and both are
|
|
// resumable, but the remedies are opposite — buy again, against top up — and a user told
|
|
// «your credit ran out» beside a balance with money on it goes and does the wrong one.
|
|
//
|
|
// `paused` WITH its reason: the contract describes PausedReason as the reason a run is paused,
|
|
// and a paused run with a null reason gives the screen nothing to say. PauseRun is the path
|
|
// that sets it; FinishRun does not. Done HERE and not inside reopen because the other caller —
|
|
// a resume — works on a run that has already ended, and pausing one of those would rewrite how
|
|
// it ended.
|
|
//
|
|
// A stop that arrived while this pass was settling outranks the pause: the store refuses and
|
|
// the run ends as what it is, stopped.
|
|
// The reason the STREAM gave wins over this path's own: overwriting a daily-ceiling halt with
|
|
// the account's reason is what made it look like an exhausted account.
|
|
//
|
|
// ⚠ Unreachable today and kept anyway, which is stated rather than implied: `restart` is only
|
|
// entered from branches that have already established the run reported no ceiling, so a landing
|
|
// on these three lines SURVIVES. It is a belt against a later path reaching here with a reason
|
|
// already set — the same shape as the ErrStopRequested answer in Resume below.
|
|
reason := pgstore.PausedRunLimitReached
|
|
if v == creditUnavailable {
|
|
reason = pgstore.PausedCreditExhausted
|
|
}
|
|
if l.PausedReason != "" {
|
|
// The reason the STREAM gave wins over this path's own: overwriting a daily-ceiling halt
|
|
// with either of ours is what made a run look like an exhausted account.
|
|
reason = l.PausedReason
|
|
}
|
|
paused, err := s.Store.PauseRun(ctx, l.RunID, l.AttemptID, reason, s.now())
|
|
if err != nil {
|
|
if errors.Is(err, pgstore.ErrStopRequested) {
|
|
s.log().InfoContext(ctx, "pause abandoned: the run was asked to stop", "run", l.RunID)
|
|
return s.finishStopped(ctx, l)
|
|
}
|
|
return err
|
|
}
|
|
if !paused {
|
|
// The run was already over, or the attempt this pass holds is no longer its live one.
|
|
// Said rather than assumed: the line below would otherwise announce a pause that did not
|
|
// happen, which is what the register row is about (PD-141).
|
|
s.log().InfoContext(ctx, "the run moved on since this pass read it; nothing to pause",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return nil
|
|
}
|
|
s.log().InfoContext(ctx, "run paused", "run", l.RunID, "reason", reason)
|
|
return nil
|
|
case deferred:
|
|
return nil
|
|
}
|
|
s.log().InfoContext(ctx, "interrupted run restarted", "run", l.RunID, "attempt", next.AttemptNo)
|
|
return s.spawnAttempt(ctx, next)
|
|
}
|
|
|
|
// verdict is what reopening a run came to. Three outcomes because the callers want different things
|
|
// from the two that are not "it started": the reconciler treats both as "nothing more this pass",
|
|
// while a user's resume must tell "your money is still being counted, ask again" from "there is
|
|
// nothing left of this run's budget".
|
|
type verdict int
|
|
|
|
const (
|
|
reopened verdict = iota
|
|
// deferred — the previous attempt's money is not settled yet. Temporary by construction: the
|
|
// settlement is retried by every sweep.
|
|
deferred
|
|
// ceilingSpent — the run has spent the ceiling it bought. There is no work this call could pay
|
|
// for, in any status, and the remedy is a NEW run.
|
|
ceilingSpent
|
|
// creditUnavailable — the run has room LEFT in its ceiling and the account cannot carry the hold
|
|
// for it. The opposite remedy: top up, and this same run continues. Folding the two together sent
|
|
// a user whose run had chapters left off to buy a run they did not need (canon §resumeRun).
|
|
creditUnavailable
|
|
)
|
|
|
|
// reopen closes an attempt and opens the next one with what is LEFT of the run's budget.
|
|
//
|
|
// It is the shared body of the reconciler's restart and of the contract's resume, and sharing it is
|
|
// the point: both are "this run continues in a new process", both must give the new attempt the
|
|
// remainder rather than the whole ceiling — one run may not spend its ceiling twice — and both must
|
|
// refuse to open a second hold while the first is still open.
|
|
// requireLive says whether the CALLER is working from a snapshot of a run it believes is still
|
|
// going. The reconciler is; a resume is not — it continues a run that has ended on purpose.
|
|
type liveness bool
|
|
|
|
const (
|
|
// fromALiveRun is the reconciler: its snapshot was taken before it did anything, and another
|
|
// sweep generation can have finished the run in between (register row PD-181's class).
|
|
fromALiveRun liveness = true
|
|
// fromAFinishedRun is the resume handle: the run has ended and re-opening it is the point.
|
|
fromAFinishedRun liveness = false
|
|
)
|
|
|
|
func (s *Service) reopen(ctx context.Context, l pgstore.LiveRun, from liveness) (pgstore.LiveRun, verdict, error) {
|
|
// Settling is allowed to DEFER — the engine's figure may not be readable — and a deferral must not
|
|
// become a second reservation. Asked rather than assumed: proceeding on an unsettled attempt holds
|
|
// the ceiling twice and strands the first hold where no later sweep looks for it.
|
|
open, err := s.Store.AttemptReservationOpen(ctx, l.RunID, l.AttemptNo)
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
if open {
|
|
s.log().WarnContext(ctx, "restart deferred: the previous attempt is not settled yet",
|
|
"run", l.RunID, "attempt", l.AttemptNo)
|
|
return pgstore.LiveRun{}, deferred, nil
|
|
}
|
|
// The budget is READ, never derived again: what this run was sold for is the hold its first
|
|
// attempt took, and everything that produced that hold can have moved since — the engine's own
|
|
// projection is re-derived on every re-cut, and the cushion above it is a deployment setting.
|
|
// Quoting the order again at today's figures would re-price a paid run on the day it continues —
|
|
// up, and the restart holds more than the buyer was ever shown; down, and the remainder goes
|
|
// negative and a run with chapters left pauses as exhausted (PD-168: with the setting doubled,
|
|
// $5.50 held for a $2.50 remainder). A run without a first hold is a broken invariant and is
|
|
// answered as an error, not as a figure.
|
|
budget, err := s.Store.RunBudget(ctx, l.RunID)
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
spent, err := s.Store.RunSpent(ctx, l.RunID)
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
remaining := budget - spent
|
|
if remaining <= 0 {
|
|
s.log().InfoContext(ctx, "the run has no budget left", "run", l.RunID)
|
|
return pgstore.LiveRun{}, ceilingSpent, nil
|
|
}
|
|
offset, err := journalSize(l.Workdir)
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
// Nil = inherit the pinned build. A different one only when an operator asked for it (row 139).
|
|
var version *string
|
|
if s.Cfg.AllowEngineVersionChange && s.Cfg.EngineBinary != "" {
|
|
version = &s.Cfg.EngineBinary
|
|
}
|
|
// The re-pass consents, granted on the USER's resume alone (P10 §3.1): a resume of a run the
|
|
// bank moved under respawns into the engine's snapshot guard, and without the flags it dies
|
|
// loudly after the money moved. The cap is the run's own full budget — funded consent, the same
|
|
// figure Start granted, read from the SAME hold: priced again at today's rate it would be the
|
|
// same figure only while the rate stood still (a projection to cap against does not exist here;
|
|
// errata 28.08-к). The reconciler's restarts (fromALiveRun) pass zeroes and change nothing:
|
|
// their run was admitted with its consents already on the row, and argv is the admission's
|
|
// decision, never a sweep's re-derivation.
|
|
resnapshot, consent := false, money.MicroUSD(0)
|
|
if from == fromAFinishedRun {
|
|
book, err := s.Store.ReadBookForRun(ctx, l.UserID, l.BookID)
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
if book.BankMoved {
|
|
resnapshot = true
|
|
consent = budget
|
|
}
|
|
}
|
|
next, err := s.Store.RestartRun(ctx, pgstore.RestartInput{
|
|
RunID: l.RunID,
|
|
AttemptID: l.AttemptID,
|
|
Resnapshot: resnapshot,
|
|
AcceptRebill: consent,
|
|
// The reconciler's own guard against a stale snapshot: if another pass finished this run
|
|
// while this one was settling, it must not be brought back to life.
|
|
OnlyIfLive: bool(from),
|
|
UserID: l.UserID,
|
|
BookID: l.BookID,
|
|
Ceiling: remaining,
|
|
Offset: offset,
|
|
EngineBinary: version,
|
|
Now: s.now(),
|
|
})
|
|
if errors.Is(err, pgstore.ErrInsufficientCredit) {
|
|
// The balance went elsewhere while this run was down. Honest state, not a failure — and NOT
|
|
// the same fact as a spent ceiling: what is left of this run is still there to continue.
|
|
s.log().InfoContext(ctx, "the run cannot be continued on the current balance", "run", l.RunID)
|
|
return pgstore.LiveRun{}, creditUnavailable, nil
|
|
}
|
|
if err != nil {
|
|
return pgstore.LiveRun{}, deferred, err
|
|
}
|
|
return next, reopened, nil
|
|
}
|
|
|
|
func (s *Service) enqueueNow(ctx context.Context, runID string) error {
|
|
if s.Queue == nil {
|
|
return nil // no queue configured: the reconciler still picks the run up on its next sweep
|
|
}
|
|
return s.Queue.EnqueueRunNow(ctx, runID)
|
|
}
|
|
|
|
// Stop is the contract's stop handle (§stopRun).
|
|
//
|
|
// What it does is record the INTENT and then ask systemd, in that order, and the order is the whole
|
|
// mechanism: the engine catches SIGTERM and exits 1, so the exit marker says the same thing for a
|
|
// stop and for a crash (register row PD-152). The row written before the signal is what tells them
|
|
// apart afterwards — and what stops the reconciler restarting a run whose owner just cancelled it.
|
|
//
|
|
// The systemd call is allowed to fail without failing the request: the intent is committed, the
|
|
// reconciler re-issues the stop on its next pass, and answering an error to a stop that WILL happen
|
|
// would invite the user to press it again.
|
|
func (s *Service) Stop(ctx context.Context, userID, runID string) (pgstore.Run, error) {
|
|
run, unit, err := s.Store.RequestStop(ctx, userID, runID, s.now())
|
|
if errors.Is(err, pgstore.ErrRunNotLive) {
|
|
// No status in the message: RequestStop answers a zero Run alongside this error, so anything
|
|
// read off it would be an empty string dressed up as a fact.
|
|
return pgstore.Run{}, ErrNotStoppable
|
|
}
|
|
if err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
s.log().InfoContext(ctx, "run stop requested", "run", runID)
|
|
if unit == "" {
|
|
// Admitted and not yet spawned. There is nothing to signal; the reconciler ends the run and
|
|
// gives the hold back on its next pass.
|
|
return run, nil
|
|
}
|
|
if err := s.Runner.Stop(ctx, unit); err != nil {
|
|
s.log().ErrorContext(ctx, "the stop was recorded but systemd did not take it; the reconciler will retry",
|
|
"run", runID, "err", err)
|
|
}
|
|
return run, nil
|
|
}
|
|
|
|
// Resume is the contract's resume handle (§resumeRun): it continues a run that was stopped.
|
|
//
|
|
// The mechanics are the reconciler's — a new attempt with what is LEFT of the run's budget — because
|
|
// a resume and a restart are the same event seen from two sides, and a second implementation of
|
|
// "give this run another process" is a second place for the money to be got wrong.
|
|
//
|
|
// What this adds is the state machine the contract describes, and its shape is dictated by that text
|
|
// rather than chosen (canon §resumeRun, the table by status):
|
|
// - a run the user stopped continues;
|
|
// - a BANK STOP continues, with the decisions AS THEY STAND — signing is one act over the whole
|
|
// bank, not a march through every row (D39.144);
|
|
// - a run paused at a LIMIT answers 409 `run_not_resumable` with `cause.code: ceiling_reached`,
|
|
// because the remedy is a NEW run with a larger limit and this call can never be it;
|
|
// - anything else — already going, finished, failed — is a 409 with no cause.
|
|
func (s *Service) Resume(ctx context.Context, userID, runID string) (pgstore.Run, error) {
|
|
// Ownership FIRST, deployment health second. The other order answers "this service cannot start
|
|
// runs" to a caller asking about a run that is not theirs and to one asking about a run that does
|
|
// not exist — which is both a worse answer and a small oracle about the shape of the deployment.
|
|
l, err := s.Store.ReadRunForResume(ctx, userID, runID)
|
|
if err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
if err := s.runnable(); err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
// The book's own serialization (bank.go): a resume must not re-open the run while the correction
|
|
// door is mid-verb on this book — the spawned engine would die on the verb's flock, a wasted
|
|
// attempt. Held through the re-open so the door's own live-run check reads a settled fact; the
|
|
// caller's context bounds the wait.
|
|
unlock, err := s.lockBook(ctx, l.BookID)
|
|
if err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
defer unlock()
|
|
// Only the book's LATEST run may be resumed. An older stopped run re-opened over a newer one
|
|
// counts the newer run's finished chapters into its own bar (its baselines predate them, and
|
|
// the clamp bounds the fraction at one, not the double count under it) — and because every
|
|
// book-scoped read resolves the run by `started_at desc`, the card and every progress frame
|
|
// would keep quoting the finished neighbour while this run burned money invisibly (workflow
|
|
// finding, P9; PD-402 keeps the read-side half). Checked under the book lock, so a Start
|
|
// admitted while this call waited cannot slip a newer row past it. The two shapes carry their
|
|
// own words: a LIVE newer run is «the book is being translated» — the same answer every door
|
|
// gives a working book — and only a finished one says «resume the latest».
|
|
latest, live, err := s.Store.LatestRun(ctx, l.BookID)
|
|
if err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
if latest != l.RunID {
|
|
if live {
|
|
return pgstore.Run{}, pgstore.ErrRunInFlight
|
|
}
|
|
return pgstore.Run{}, fmt.Errorf("%w: a newer run of this book exists, and the book's screens follow that one", ErrNotResumable)
|
|
}
|
|
if l.OrderedChapters == 0 {
|
|
// A re-pass is bought again, not resumed (the adversarial pass's K4, which sealed this door).
|
|
// Nothing needs resuming: an interrupted re-pass leaves the fact standing (only a READY
|
|
// resnapshot run retires it), so the purchase is simply available again — and a user's
|
|
// resume is a second purchase under the guise of a continuation. The reconciler's own restart
|
|
// of a re-pass that a reboot interrupted is a different act: the same purchase continuing on
|
|
// what is left of its hold, which reopen reads from that hold like any other run's.
|
|
return pgstore.Run{}, fmt.Errorf("%w: a re-pass is bought again rather than resumed", ErrNotResumable)
|
|
}
|
|
switch l.Status {
|
|
case "stopped":
|
|
case "awaiting_bank":
|
|
// A bank stop is lifted by this call at ANY state of the decisions. ⚠ The gate that used to
|
|
// stand here — "the stop clears only on a complete set of decisions" — was an invention of the
|
|
// contract line that outlived its own basis: the engine has always continued with an unsigned
|
|
// bank by default, carrying untouched terms forward marked unverified (mining.go, D39.42 п.3),
|
|
// and the owner's model of 16.08 is one OK over the whole bank. D39.144 removed it from the
|
|
// canon; this is the built half being dismantled with it.
|
|
case "paused":
|
|
// ⚠ EVERY pause answers 409 `ceiling_reached`, not only the engine's daily one. The limit
|
|
// travels with the START of a run and nothing changes it afterwards, so this call cannot move
|
|
// a run that stopped at one — and answering 202 with the run in the state it was already in is
|
|
// exactly the silent no-op the contract now forbids (canon §resumeRun; the client has no way
|
|
// to tell that success from a real continuation). The remedy is a NEW run with a larger
|
|
// ceiling, which is legal from ANY paused book including one whose reason is null.
|
|
return pgstore.Run{}, ErrCeilingReached
|
|
default:
|
|
return pgstore.Run{}, fmt.Errorf("%w: it is %s", ErrNotResumable, l.Status)
|
|
}
|
|
next, v, err := s.reopen(ctx, l, fromAFinishedRun)
|
|
if errors.Is(err, pgstore.ErrStopRequested) {
|
|
// Cannot happen through this path today — a resume works on a run that has ENDED and the guard
|
|
// only fires on a live one — but the answer is written rather than left to fall through as an
|
|
// internal error if some later path reaches it.
|
|
return pgstore.Run{}, fmt.Errorf("%w: it has been asked to stop", ErrNotResumable)
|
|
}
|
|
if errors.Is(err, pgstore.ErrNoRun) {
|
|
// Another caller re-opened this run first — two clicks on the same button reach here together,
|
|
// and the second loses the race for attempt N+1 on the unique index. Its answer is the WORK OF
|
|
// THE FIRST: the run is continuing, so reporting "no such run" would be both false and
|
|
// alarming, and re-opening again would take a second hold.
|
|
return s.Store.ReadRun(ctx, userID, runID)
|
|
}
|
|
if err != nil {
|
|
return pgstore.Run{}, err
|
|
}
|
|
switch v {
|
|
case deferred:
|
|
// The previous attempt's money is still being resolved. A second hold now would reserve the
|
|
// ceiling twice, so the answer is "not yet" and the sweep clears it.
|
|
return pgstore.Run{}, fmt.Errorf("%w: its previous attempt is still being settled", ErrNotResumable)
|
|
case ceilingSpent:
|
|
// The run bought a ceiling and spent it, so there is nothing this call could pay for — in ANY
|
|
// status, and however much credit the account holds. The remedy is a NEW run, and the state
|
|
// the run is in is not rewritten on the way out: a run the user stopped stays stopped.
|
|
//
|
|
// It used to answer 202 with the run unchanged, which is a success a client cannot tell from
|
|
// a continuation (PD-282). The canon now says the opposite in as many words: a 202 means work
|
|
// was actually re-opened.
|
|
return pgstore.Run{}, ErrCeilingReached
|
|
case creditUnavailable:
|
|
// The run has room LEFT and the account cannot cover it. The remedy is to top up — after
|
|
// which this same call continues this same run — so it must not be answered `ceiling_reached`,
|
|
// which would send the user off to buy a run they do not need.
|
|
return pgstore.Run{}, ErrCreditUnavailable
|
|
}
|
|
s.log().InfoContext(ctx, "run resumed", "run", runID, "attempt", next.AttemptNo)
|
|
// QUEUED, not spawned here. Starting the engine inline would put two slow things inside a request
|
|
// that answers 202: reading the book's meter costs seconds of the engine's CPU, and creating the
|
|
// unit is a round trip to systemd. The admission path queues for the same reason, and the
|
|
// reconciler is the backstop if the entry is lost.
|
|
if err := s.enqueueNow(ctx, next.RunID); err != nil {
|
|
// The run IS re-opened and the money IS held: the row says `translating` and the reconciler
|
|
// picks up an attempt with no unit on its next pass, exactly as it does after an admission.
|
|
// Reporting a failure here would tell the user nothing had started while the row said
|
|
// otherwise.
|
|
s.log().ErrorContext(ctx, "the resumed run could not be queued; the reconciler will retry",
|
|
"run", runID, "err", err)
|
|
}
|
|
return s.Store.ReadRun(ctx, userID, runID)
|
|
}
|
|
|
|
// attemptSpend turns the engine's figure into what THIS attempt cost.
|
|
//
|
|
// ⚠ The engine's `committed_usd` is a lifetime total for the BOOK — literally
|
|
// `SELECT COALESCE(SUM(committed_usd),0) FROM spend WHERE book_id = ?`
|
|
// (backend/internal/store/ledger.go at HEAD, reached through pipeline/status.go) — across every run
|
|
// and every day. Settling with it verbatim charges an account for all of the book's earlier runs
|
|
// again, every time; the overcharge is bounded by the hold, and the ledger then reads "capped at the
|
|
// hold", which looks like an engine overspend rather than the platform's own arithmetic. So each
|
|
// attempt records where the meter stood when it started and owes the DIFFERENCE.
|
|
//
|
|
// Clamped at zero: a total that went DOWN is not a refund, it is a book whose project database was
|
|
// replaced, and paying an account for that is not something this code may decide.
|
|
func attemptSpend(l pgstore.LiveRun, committed money.MicroUSD) (money.MicroUSD, error) {
|
|
if l.SpendBaseline == nil {
|
|
return 0, errors.New("runs: the attempt has no spend baseline")
|
|
}
|
|
if committed < *l.SpendBaseline {
|
|
return 0, nil
|
|
}
|
|
return committed - *l.SpendBaseline, nil
|
|
}
|