From 3e99eeedff7cc5b5665c75076cdffd36491d7e0a Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Thu, 10 Sep 2026 06:44:01 +0800 Subject: [PATCH 1/8] feat(office): bound run history with a scheduled retention sweep Adds a scheduled sweep, independent of the 5s Office tick, that bounds office_routine_runs, runs, and their satellite tables (run_events, office_run_route_attempts, office_run_skills). History is classified by status alone; live-state rows are never deleted. Deletion is age-based (default 30 days) with a per-owner retention floor, batched and oldest-first, guarded by a session-scoped pg_try_advisory_lock on PostgreSQL for cross-process exclusivity. Ships with a per-table preview before first deletion, health warnings, a GET/PUT /api/v1/system/retention settings API, and a frontend settings card. Co-Authored-By: Claude Sonnet 5 --- apps/backend/internal/backendapp/helpers.go | 19 + apps/backend/internal/backendapp/main.go | 11 + apps/backend/internal/backendapp/types.go | 7 + .../repository/sqlite/base_migrations.go | 24 + .../sqlite/retention_indexes_postgres_test.go | 51 ++ .../sqlite/retention_indexes_test.go | 51 ++ .../internal/office/retention/census.go | 161 +++++ .../office/retention/census_json_test.go | 101 ++++ .../internal/office/retention/census_test.go | 133 +++++ .../internal/office/retention/handler.go | 108 ++++ .../internal/office/retention/handler_test.go | 285 +++++++++ .../internal/office/retention/health.go | 229 +++++++ .../internal/office/retention/health_test.go | 364 ++++++++++++ .../backend/internal/office/retention/lock.go | 109 ++++ .../office/retention/lock_postgres_test.go | 141 +++++ .../internal/office/retention/metrics_vars.go | 55 ++ .../office/retention/metrics_vars_test.go | 116 ++++ .../internal/office/retention/policy.go | 59 ++ .../internal/office/retention/policy_test.go | 98 +++ .../office/retention/preview_marker.go | 117 ++++ .../office/retention/preview_marker_test.go | 117 ++++ .../internal/office/retention/runtime.go | 61 ++ .../internal/office/retention/runtime_test.go | 79 +++ .../internal/office/retention/scheduler.go | 174 ++++++ .../office/retention/scheduler_test.go | 272 +++++++++ .../office/retention/settings_store.go | 91 +++ .../office/retention/settings_store_test.go | 154 +++++ .../office/retention/settings_wire.go | 169 ++++++ .../internal/office/retention/store.go | 380 ++++++++++++ .../office/retention/store_census_test.go | 159 +++++ .../internal/office/retention/store_test.go | 428 ++++++++++++++ .../internal/office/retention/sweep.go | 322 ++++++++++ .../office/retention/sweep_postgres_test.go | 201 +++++++ .../internal/office/retention/sweep_test.go | 325 ++++++++++ .../internal/office/retention/types.go | 146 +++++ .../internal/office/retention/types_test.go | 94 +++ .../service/scheduler_checkout_error_test.go | 13 +- .../settings/system/data-logs-settings.tsx | 9 + .../system/retention-settings-card.test.tsx | 196 ++++++ .../system/retention-settings-card.tsx | 551 +++++++++++++++++ .../domains/system/use-retention-settings.ts | 49 ++ apps/web/lib/api/domains/system-api.test.ts | 46 ++ apps/web/lib/api/domains/system-api.ts | 25 + .../lib/settings-discovery/catalog/system.ts | 15 +- .../state/slices/system/system-slice.test.ts | 24 + .../lib/state/slices/system/system-slice.ts | 5 + apps/web/lib/state/slices/system/types.ts | 3 + apps/web/lib/types/system.ts | 67 +++ apps/web/src/locales/en/system.json | 44 ++ apps/web/src/locales/pseudo/system.json | 44 ++ apps/web/src/locales/pt-pt/system.json | 44 ++ apps/web/src/locales/zh-cn/system.json | 44 ++ apps/web/src/locales/zh-hk/system.json | 44 ++ apps/web/src/locales/zh-tw/system.json | 44 ++ .../run-history-retention-operations.md | 214 +++++++ .../requirements/run-history-retention.md | 316 ++++++++++ .../run-history-retention-operations.md | 376 ++++++++++++ .../system-design/run-history-retention.md | 557 ++++++++++++++++++ 58 files changed, 8137 insertions(+), 4 deletions(-) create mode 100644 apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go create mode 100644 apps/backend/internal/office/repository/sqlite/retention_indexes_test.go create mode 100644 apps/backend/internal/office/retention/census.go create mode 100644 apps/backend/internal/office/retention/census_json_test.go create mode 100644 apps/backend/internal/office/retention/census_test.go create mode 100644 apps/backend/internal/office/retention/handler.go create mode 100644 apps/backend/internal/office/retention/handler_test.go create mode 100644 apps/backend/internal/office/retention/health.go create mode 100644 apps/backend/internal/office/retention/health_test.go create mode 100644 apps/backend/internal/office/retention/lock.go create mode 100644 apps/backend/internal/office/retention/lock_postgres_test.go create mode 100644 apps/backend/internal/office/retention/metrics_vars.go create mode 100644 apps/backend/internal/office/retention/metrics_vars_test.go create mode 100644 apps/backend/internal/office/retention/policy.go create mode 100644 apps/backend/internal/office/retention/policy_test.go create mode 100644 apps/backend/internal/office/retention/preview_marker.go create mode 100644 apps/backend/internal/office/retention/preview_marker_test.go create mode 100644 apps/backend/internal/office/retention/runtime.go create mode 100644 apps/backend/internal/office/retention/runtime_test.go create mode 100644 apps/backend/internal/office/retention/scheduler.go create mode 100644 apps/backend/internal/office/retention/scheduler_test.go create mode 100644 apps/backend/internal/office/retention/settings_store.go create mode 100644 apps/backend/internal/office/retention/settings_store_test.go create mode 100644 apps/backend/internal/office/retention/settings_wire.go create mode 100644 apps/backend/internal/office/retention/store.go create mode 100644 apps/backend/internal/office/retention/store_census_test.go create mode 100644 apps/backend/internal/office/retention/store_test.go create mode 100644 apps/backend/internal/office/retention/sweep.go create mode 100644 apps/backend/internal/office/retention/sweep_postgres_test.go create mode 100644 apps/backend/internal/office/retention/sweep_test.go create mode 100644 apps/backend/internal/office/retention/types.go create mode 100644 apps/backend/internal/office/retention/types_test.go create mode 100644 apps/web/components/settings/system/retention-settings-card.test.tsx create mode 100644 apps/web/components/settings/system/retention-settings-card.tsx create mode 100644 apps/web/hooks/domains/system/use-retention-settings.ts create mode 100644 docs/specs/office/requirements/run-history-retention-operations.md create mode 100644 docs/specs/office/requirements/run-history-retention.md create mode 100644 docs/specs/office/system-design/run-history-retention-operations.md create mode 100644 docs/specs/office/system-design/run-history-retention.md diff --git a/apps/backend/internal/backendapp/helpers.go b/apps/backend/internal/backendapp/helpers.go index 859ad57c127..fb28344d202 100644 --- a/apps/backend/internal/backendapp/helpers.go +++ b/apps/backend/internal/backendapp/helpers.go @@ -65,6 +65,7 @@ import ( notificationhandlers "github.com/kandev/kandev/internal/notifications/handlers" officeagents "github.com/kandev/kandev/internal/office/agents" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officetestharness "github.com/kandev/kandev/internal/office/testharness" "github.com/kandev/kandev/internal/orchestrator" "github.com/kandev/kandev/internal/org" @@ -1521,6 +1522,7 @@ func registerSecondaryRoutes( registerHealthRoutes(p) registerSystemRoutes(p) + registerRetentionRoutes(p) if p.runtimeFlagsSvc != nil { runtimeflags.RegisterRoutes(p.router, p.runtimeFlagsSvc) } @@ -1722,6 +1724,20 @@ func registerSystemRoutes(p routeParams) { p.systemSvc.RegisterRoutes(p.router, p.log) } +// registerRetentionRoutes mounts GET/PUT /api/v1/system/retention. It is a +// separate admin-scoped group from systemSvc's own /api/v1/system group +// (rather than a field on system.Service) because internal/office/retention +// cannot be imported by internal/system without inverting the existing +// system -> office dependency direction; gin allows two RouterGroups to +// share a path prefix as long as no route collides, and none does here. +func registerRetentionRoutes(p routeParams) { + if p.services == nil || p.services.Retention == nil { + return + } + admin := p.router.Group("/api/v1/system", authz.RequireOrgScope(authz.ScopeOrgSettingsManage)) + retention.RegisterRoutes(admin, p.services.Retention.Handler) +} + // registerHealthRoutes sets up the system health endpoint with all health checkers. func registerHealthRoutes(p routeParams) { var githubProvider health.GitHubStatusProvider @@ -1747,6 +1763,9 @@ func registerHealthRoutes(p routeParams) { if p.systemSvc != nil && p.systemSvc.StorageRuntime != nil { checkers = append(checkers, p.systemSvc.StorageRuntime) } + if p.services != nil && p.services.Retention != nil { + checkers = append(checkers, p.services.Retention.Checker) + } healthSvc := health.NewService(p.log, checkers...) health.RegisterRoutes(p.router, healthSvc, p.log) } diff --git a/apps/backend/internal/backendapp/main.go b/apps/backend/internal/backendapp/main.go index 845bc80d67f..7ef9ad79846 100644 --- a/apps/backend/internal/backendapp/main.go +++ b/apps/backend/internal/backendapp/main.go @@ -100,6 +100,7 @@ import ( officepause "github.com/kandev/kandev/internal/office/pause" officeprojects "github.com/kandev/kandev/internal/office/projects" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officeroutines "github.com/kandev/kandev/internal/office/routines" "github.com/kandev/kandev/internal/office/routing" officescheduler "github.com/kandev/kandev/internal/office/scheduler" @@ -1181,6 +1182,16 @@ func startGatewayAndServe( }) systemSvc.Storage = storageComposition.handler systemSvc.StorageRuntime = storageComposition.runtime + + // Office run history retention: bounds office_routine_runs, runs, and + // their satellites on its own interval, separate from the 5s Office + // tick. Kept regardless of the Office feature flag — see Services.Retention. + services.Retention = retention.NewRuntime(dbPool, repos.SystemSettings, + func(message string, err error) { log.Error(message, zap.Error(err)) }) + if err := services.Retention.Start(ctx); err != nil { + log.Warn("office run retention scheduler failed to start", zap.Error(err)) + } + addCleanup(func() error { services.Retention.Stop(); return nil }) if systemSvc.LogBundles != nil { systemSvc.LogBundles.SetNotifier(gateway.Hub) systemSvc.LogBundles.SetSessionProvider(newDiagnosticSessionProvider(services.Task)) diff --git a/apps/backend/internal/backendapp/types.go b/apps/backend/internal/backendapp/types.go index 44107a9285f..84878a81fd7 100644 --- a/apps/backend/internal/backendapp/types.go +++ b/apps/backend/internal/backendapp/types.go @@ -25,6 +25,7 @@ import ( notificationstore "github.com/kandev/kandev/internal/notifications/store" office "github.com/kandev/kandev/internal/office" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officeservice "github.com/kandev/kandev/internal/office/service" "github.com/kandev/kandev/internal/org" "github.com/kandev/kandev/internal/orgunit" @@ -112,6 +113,12 @@ type Services struct { // WorktreeMgr is the worktree manager. Exposed here so the install-wide // storage-maintenance composition can reach it for workspace cleanup. WorktreeMgr *worktree.Manager + // Retention owns the office_routine_runs/runs history sweep scheduler, + // its HTTP surface, and its health checker. Kept regardless of the + // Office feature flag, matching every other required-schema owner: rows + // written while Office was enabled still need bounding after it is + // turned off. + Retention *retention.Runtime // Terminal is the first-class user-terminal service (rename, park, etc.). // Wired into the gateway once lifecycle.Manager is up so the PTY backend // is available. diff --git a/apps/backend/internal/office/repository/sqlite/base_migrations.go b/apps/backend/internal/office/repository/sqlite/base_migrations.go index 75f2b6c6e99..5f7fca3f3a2 100644 --- a/apps/backend/internal/office/repository/sqlite/base_migrations.go +++ b/apps/backend/internal/office/repository/sqlite/base_migrations.go @@ -66,6 +66,9 @@ func (r *Repository) runMigrations() error { r.migrateBudgetPolicyRevision() r.migrateWorkspacePauseSkipAttribution() r.migrateLoopLivenessCausationID() + if err := r.migrateRetentionIndexes(); err != nil { + return err + } if err := r.migrate.Err(); err != nil { return err } @@ -141,6 +144,27 @@ func (r *Repository) migrateLoopLivenessCausationID() { ON runs(causation_id) WHERE causation_id != ''`) } +// migrateRetentionIndexes adds the two indexes the run-history retention +// sweep depends on (docs/specs/office/system-design/run-history-retention.md +// "Indexes to add"). Both are expression indexes over the same +// COALESCE(...) the sweep both filters and orders by; a plain-column index +// on the nullable completion column would serve neither the WHERE clause +// nor the ORDER BY the sweep actually issues, on either engine. +func (r *Repository) migrateRetentionIndexes() error { + if err := r.migrate.Apply( + "idx_office_routine_runs_retention", + `CREATE INDEX IF NOT EXISTS idx_office_routine_runs_retention + ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)`, + ); err != nil { + return err + } + return r.migrate.Apply( + "idx_runs_retention", + `CREATE INDEX IF NOT EXISTS idx_runs_retention + ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)`, + ) +} + // migrateContinuationScope adds runs.continuation_scope for databases // created before WO-16's claim-time scope persistence. Existing rows receive // a scope from their stored context snapshot so queued or claimed taskless diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go new file mode 100644 index 00000000000..35f3bd2362e --- /dev/null +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go @@ -0,0 +1,51 @@ +package sqlite_test + +import ( + "testing" + + "github.com/kandev/kandev/internal/office/repository/sqlite" + taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite" + "github.com/kandev/kandev/internal/testutil" +) + +// TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe is the PostgreSQL +// half of TestRetentionIndexes_CreatedFreshAndReplaySafe: the two +// expression indexes the retention sweep depends on must exist there too, +// with identical CREATE INDEX IF NOT EXISTS replay safety. Skips unless +// KANDEV_TEST_POSTGRES_DSN is set. +func TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + conn := testutil.OpenIsolatedPostgres(t, dsn) + + // tasks is created by the task repository's schema init, mirroring + // production boot order (see child_summaries_postgres_test.go). + if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init task repo: %v", err) + } + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("fresh NewWithDB: %v", err) + } + assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") + assertPostgresIndexExists(t, conn, "idx_runs_retention") + + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("replay NewWithDB: %v", err) + } + assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") + assertPostgresIndexExists(t, conn, "idx_runs_retention") +} + +func assertPostgresIndexExists(t *testing.T, conn interface { + Get(dest interface{}, query string, args ...interface{}) error +}, name string) { + t.Helper() + var count int + if err := conn.Get(&count, + `SELECT COUNT(*) FROM pg_indexes WHERE indexname = $1`, name, + ); err != nil { + t.Fatalf("query pg_indexes for %s: %v", name, err) + } + if count != 1 { + t.Fatalf("index %s: found %d, want 1", name, count) + } +} diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go new file mode 100644 index 00000000000..e95919f9134 --- /dev/null +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go @@ -0,0 +1,51 @@ +package sqlite_test + +import ( + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// TestRetentionIndexes_CreatedFreshAndReplaySafe proves the two indexes the +// retention sweep depends on (idx_office_routine_runs_retention, +// idx_runs_retention) exist after a fresh boot and that re-running schema +// init against the same database (the upgrade-path replay) is a no-op, not +// an error — CREATE INDEX IF NOT EXISTS over the same COALESCE(...) +// expression both paths must produce identically. +func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { + conn, err := sqlx.Open("sqlite3", ":memory:") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("fresh NewWithDB: %v", err) + } + assertIndexExists(t, conn, "idx_office_routine_runs_retention") + assertIndexExists(t, conn, "idx_runs_retention") + + // Replay: schema init against the same, already-initialized database. + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("replay NewWithDB: %v", err) + } + assertIndexExists(t, conn, "idx_office_routine_runs_retention") + assertIndexExists(t, conn, "idx_runs_retention") +} + +func assertIndexExists(t *testing.T, conn *sqlx.DB, name string) { + t.Helper() + var count int + if err := conn.Get(&count, + `SELECT COUNT(*) FROM sqlite_master WHERE type = 'index' AND name = ?`, name, + ); err != nil { + t.Fatalf("query sqlite_master for %s: %v", name, err) + } + if count != 1 { + t.Fatalf("index %s: found %d, want 1", name, count) + } +} diff --git a/apps/backend/internal/office/retention/census.go b/apps/backend/internal/office/retention/census.go new file mode 100644 index 00000000000..9a8fa18d575 --- /dev/null +++ b/apps/backend/internal/office/retention/census.go @@ -0,0 +1,161 @@ +package retention + +import ( + "encoding/json" + "fmt" + "sort" + "sync" + "time" +) + +// CensusState is the tri-state freshness of a thresholded table's retained +// count (AC-OFFICE-RUN-HISTORY-RETENTION-003.11): zero is a real +// measurement, so "nobody has counted yet" cannot be represented as a +// count of zero. +type CensusState int + +const ( + // CensusNotComputed is the state before the first evaluation ever + // succeeds for a table. AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and + // -004.8 both require the surface to render this honestly rather than + // as a zero count. + CensusNotComputed CensusState = iota + // CensusFresh means RetainedCount reflects the most recent evaluation, + // which succeeded. + CensusFresh + // CensusStale means the most recent evaluation failed, and the fields + // below are carried over unchanged from the last one that succeeded + // (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's "keeps the last successful + // counts, reports them stale" — extended to the unknown-status warning + // riding the same query, since neither the requirement nor the design + // says the two field groups should diverge on a failed evaluation). + CensusStale +) + +var censusStateNames = map[CensusState]string{ + CensusNotComputed: "not_computed", + CensusFresh: "fresh", + CensusStale: "stale", +} + +// MarshalJSON renders the state as its stable wire name rather than the +// underlying int, so an HTTP consumer never has to hardcode 0/1/2. +func (s CensusState) MarshalJSON() ([]byte, error) { + name, ok := censusStateNames[s] + if !ok { + return nil, fmt.Errorf("retention: unknown census state %d", s) + } + return json.Marshal(name) +} + +// UnmarshalJSON accepts only the names MarshalJSON produces. +func (s *CensusState) UnmarshalJSON(data []byte) error { + var name string + if err := json.Unmarshal(data, &name); err != nil { + return err + } + for state, candidate := range censusStateNames { + if candidate == name { + *s = state + return nil + } + } + return fmt.Errorf("retention: unknown census state %q", name) +} + +// TableCensus is one thresholded table's retained-count evaluation. +type TableCensus struct { + State CensusState `json:"state"` + RetainedCount int64 `json:"retained_count"` + AsOf time.Time `json:"as_of"` + UnknownStatuses []string `json:"unknown_statuses,omitempty"` // ascending; only populated for status-bearing tables + TopRoutineID string `json:"top_routine_id,omitempty"` // office_routine_runs only; empty when not applicable + TopRoutineShare float64 `json:"top_routine_share,omitempty"` // top routine's retained rows / table's retained count +} + +// RetainedCounts holds the current census result for every thresholded +// table (AC-OFFICE-RUN-HISTORY-RETENTION-003.11). +type RetainedCounts struct { + OfficeRoutineRuns TableCensus `json:"office_routine_runs"` + Runs TableCensus `json:"runs"` + RunEvents TableCensus `json:"run_events"` +} + +// summarizeStatusCensus turns a status->count breakdown into a retained +// count (the sum across every status, since "retained" is the table's +// current row count) and the sorted list of statuses belonging to neither +// the history nor the live-state set (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +func summarizeStatusCensus(counts map[string]int64, history, live []string) (retained int64, unknown []string) { + known := make(map[string]bool, len(history)+len(live)) + for _, s := range history { + known[s] = true + } + for _, s := range live { + known[s] = true + } + for status, count := range counts { + retained += count + if !known[status] { + unknown = append(unknown, status) + } + } + sort.Strings(unknown) + return retained, unknown +} + +// CensusTracker holds the latest RetainedCounts in memory, applying the +// tri-state rule per table: a successful evaluation replaces a table's +// entry and marks it fresh; a failed evaluation leaves an already-computed +// entry in place and marks it stale, or leaves a never-computed entry as +// not-yet-computed. One table's failure never touches another table's +// entry. Safe for concurrent use: Record* is called from the scheduler +// goroutine and Snapshot from HTTP handlers. +type CensusTracker struct { + mu sync.Mutex + counts RetainedCounts +} + +// NewCensusTracker returns a tracker with every table not yet computed. +func NewCensusTracker() *CensusTracker { + return &CensusTracker{} +} + +// Snapshot returns the current RetainedCounts. +func (t *CensusTracker) Snapshot() RetainedCounts { + t.mu.Lock() + defer t.mu.Unlock() + return t.counts +} + +// RecordRoutineRuns applies an office_routine_runs evaluation outcome. +func (t *CensusTracker) RecordRoutineRuns(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.OfficeRoutineRuns = applyCensusResult(t.counts.OfficeRoutineRuns, fresh, err) +} + +// RecordRuns applies a runs evaluation outcome. +func (t *CensusTracker) RecordRuns(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.Runs = applyCensusResult(t.counts.Runs, fresh, err) +} + +// RecordRunEvents applies a run_events evaluation outcome. +func (t *CensusTracker) RecordRunEvents(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.RunEvents = applyCensusResult(t.counts.RunEvents, fresh, err) +} + +func applyCensusResult(prev, fresh TableCensus, err error) TableCensus { + if err != nil { + if prev.State == CensusNotComputed { + return prev + } + prev.State = CensusStale + return prev + } + fresh.State = CensusFresh + return fresh +} diff --git a/apps/backend/internal/office/retention/census_json_test.go b/apps/backend/internal/office/retention/census_json_test.go new file mode 100644 index 00000000000..7590ccf1ca5 --- /dev/null +++ b/apps/backend/internal/office/retention/census_json_test.go @@ -0,0 +1,101 @@ +package retention + +import ( + "encoding/json" + "testing" +) + +func TestCensusState_MarshalJSON_UsesStableNames(t *testing.T) { + cases := map[CensusState]string{ + CensusNotComputed: `"not_computed"`, + CensusFresh: `"fresh"`, + CensusStale: `"stale"`, + } + for state, want := range cases { + got, err := json.Marshal(state) + if err != nil { + t.Fatalf("Marshal(%v): %v", state, err) + } + if string(got) != want { + t.Fatalf("Marshal(%v) = %s, want %s", state, got, want) + } + } +} + +func TestCensusState_UnmarshalJSON_RoundTrips(t *testing.T) { + for _, state := range []CensusState{CensusNotComputed, CensusFresh, CensusStale} { + encoded, err := json.Marshal(state) + if err != nil { + t.Fatalf("Marshal(%v): %v", state, err) + } + var decoded CensusState + if err := json.Unmarshal(encoded, &decoded); err != nil { + t.Fatalf("Unmarshal(%s): %v", encoded, err) + } + if decoded != state { + t.Fatalf("round trip %v -> %s -> %v", state, encoded, decoded) + } + } +} + +func TestCensusState_UnmarshalJSON_RejectsUnknownName(t *testing.T) { + var state CensusState + if err := json.Unmarshal([]byte(`"bogus"`), &state); err == nil { + t.Fatal("expected an error for an unrecognized census state name") + } +} + +func TestRetainedCounts_JSONUsesSnakeCaseFieldNames(t *testing.T) { + counts := RetainedCounts{ + OfficeRoutineRuns: TableCensus{State: CensusFresh, RetainedCount: 3}, + } + encoded, err := json.Marshal(counts) + if err != nil { + t.Fatalf("Marshal: %v", err) + } + var raw map[string]json.RawMessage + if err := json.Unmarshal(encoded, &raw); err != nil { + t.Fatalf("Unmarshal into map: %v", err) + } + for _, key := range []string{"office_routine_runs", "runs", "run_events"} { + if _, ok := raw[key]; !ok { + t.Fatalf("RetainedCounts JSON missing key %q; got %s", key, encoded) + } + } +} + +func TestLastSweep_JSONUsesSnakeCaseFieldNames(t *testing.T) { + sweep := LastSweep{ + OfficeRoutineRuns: SweptTableResult{ + TableSweepResult: TableSweepResult{Deleted: 5, Backlog: true}, + Previewed: true, + WouldDelete: 7, + }, + } + encoded, err := json.Marshal(sweep) + if err != nil { + t.Fatalf("Marshal: %v", err) + } + var raw map[string]json.RawMessage + if err := json.Unmarshal(encoded, &raw); err != nil { + t.Fatalf("Unmarshal into map: %v", err) + } + for _, key := range []string{ + "started_at", "finished_at", "office_routine_runs", "runs", + "run_events", "route_attempts", "run_skills", + } { + if _, ok := raw[key]; !ok { + t.Fatalf("LastSweep JSON missing key %q; got %s", key, encoded) + } + } + + var routineRuns map[string]json.RawMessage + if err := json.Unmarshal(raw["office_routine_runs"], &routineRuns); err != nil { + t.Fatalf("Unmarshal office_routine_runs: %v", err) + } + for _, key := range []string{"deleted", "backlog", "error", "previewed", "would_delete"} { + if _, ok := routineRuns[key]; !ok { + t.Fatalf("SweptTableResult JSON missing key %q; got %s", key, raw["office_routine_runs"]) + } + } +} diff --git a/apps/backend/internal/office/retention/census_test.go b/apps/backend/internal/office/retention/census_test.go new file mode 100644 index 00000000000..0f82d9b60f1 --- /dev/null +++ b/apps/backend/internal/office/retention/census_test.go @@ -0,0 +1,133 @@ +package retention + +import ( + "errors" + "reflect" + "testing" + "time" +) + +func TestSummarizeStatusCensus_SumsAllStatusesRegardlessOfClass(t *testing.T) { + counts := map[string]int64{ + "done": 3, + "received": 2, + } + retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + if retained != 5 { + t.Fatalf("retained = %d, want 5", retained) + } + if len(unknown) != 0 { + t.Fatalf("unknown = %v, want none", unknown) + } +} + +func TestSummarizeStatusCensus_DetectsUnknownStatusesSortedAscending(t *testing.T) { + counts := map[string]int64{ + "done": 1, + "zeta": 1, + "alpha": 1, + "skipped": 1, + } + retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + if retained != 4 { + t.Fatalf("retained = %d, want 4", retained) + } + want := []string{"alpha", "zeta"} + if !reflect.DeepEqual(unknown, want) { + t.Fatalf("unknown = %v, want %v", unknown, want) + } +} + +func TestSummarizeStatusCensus_EmptyTableIsZeroNotUnknown(t *testing.T) { + retained, unknown := summarizeStatusCensus(map[string]int64{}, RunHistoryStatuses, RunLiveStatuses) + if retained != 0 { + t.Fatalf("retained = %d, want 0", retained) + } + if unknown != nil { + t.Fatalf("unknown = %v, want nil", unknown) + } +} + +func TestCensusTracker_SnapshotStartsNotComputedForEveryTable(t *testing.T) { + tracker := NewCensusTracker() + snap := tracker.Snapshot() + for name, c := range map[string]TableCensus{ + "office_routine_runs": snap.OfficeRoutineRuns, + "runs": snap.Runs, + "run_events": snap.RunEvents, + } { + if c.State != CensusNotComputed { + t.Fatalf("%s: state = %v, want CensusNotComputed", name, c.State) + } + } +} + +func TestCensusTracker_SuccessfulEvaluationIsFresh(t *testing.T) { + tracker := NewCensusTracker() + now := time.Now().UTC() + tracker.RecordRuns(TableCensus{RetainedCount: 42, AsOf: now}, nil) + + got := tracker.Snapshot().Runs + if got.State != CensusFresh { + t.Fatalf("state = %v, want CensusFresh", got.State) + } + if got.RetainedCount != 42 { + t.Fatalf("retained = %d, want 42", got.RetainedCount) + } + if !got.AsOf.Equal(now) { + t.Fatalf("asOf = %v, want %v", got.AsOf, now) + } +} + +func TestCensusTracker_FailedEvaluationWithNoPriorSuccessStaysNotComputed(t *testing.T) { + tracker := NewCensusTracker() + tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom")) + + got := tracker.Snapshot().OfficeRoutineRuns + if got.State != CensusNotComputed { + t.Fatalf("state = %v, want CensusNotComputed", got.State) + } +} + +func TestCensusTracker_FailedEvaluationAfterSuccessKeepsLastCountsMarkedStale(t *testing.T) { + tracker := NewCensusTracker() + first := TableCensus{ + RetainedCount: 100, + AsOf: time.Now().UTC(), + UnknownStatuses: []string{"weird"}, + TopRoutineID: "r-1", + TopRoutineShare: 0.5, + } + tracker.RecordRoutineRuns(first, nil) + + tracker.RecordRoutineRuns(TableCensus{}, errors.New("query failed")) + + got := tracker.Snapshot().OfficeRoutineRuns + if got.State != CensusStale { + t.Fatalf("state = %v, want CensusStale", got.State) + } + if got.RetainedCount != 100 { + t.Fatalf("retained = %d, want 100 (carried over)", got.RetainedCount) + } + if !reflect.DeepEqual(got.UnknownStatuses, []string{"weird"}) { + t.Fatalf("unknownStatuses = %v, want carried over", got.UnknownStatuses) + } + if got.TopRoutineID != "r-1" || got.TopRoutineShare != 0.5 { + t.Fatalf("top routine attribution not carried over: %+v", got) + } +} + +func TestCensusTracker_OneTableFailureDoesNotTouchAnother(t *testing.T) { + tracker := NewCensusTracker() + tracker.RecordRuns(TableCensus{RetainedCount: 7, AsOf: time.Now().UTC()}, nil) + + tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom")) + + snap := tracker.Snapshot() + if snap.Runs.State != CensusFresh || snap.Runs.RetainedCount != 7 { + t.Fatalf("runs entry disturbed by routine_runs failure: %+v", snap.Runs) + } + if snap.OfficeRoutineRuns.State != CensusNotComputed { + t.Fatalf("routine_runs state = %v, want CensusNotComputed", snap.OfficeRoutineRuns.State) + } +} diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go new file mode 100644 index 00000000000..0b232301fac --- /dev/null +++ b/apps/backend/internal/office/retention/handler.go @@ -0,0 +1,108 @@ +package retention + +import ( + "errors" + "net/http" + "time" + + "github.com/gin-gonic/gin" +) + +const responseErrorKey = "error" + +// HandlerConfig wires the HTTP surface to the package's own stores. +type HandlerConfig struct { + SettingsStore *SettingsStore + Sweeper *Sweeper + // OnSettingsChanged, when set, is called with the normalized document + // after a successful PUT so the running scheduler re-arms its timers + // from the new settings without a restart (AC-004.5). + OnSettingsChanged func(Settings) + LogError func(string, error) +} + +// Handler serves GET/PUT /api/v1/system/retention. +type Handler struct { + config HandlerConfig +} + +// NewHandler wires a Handler to its dependencies. +func NewHandler(config HandlerConfig) *Handler { + return &Handler{config: config} +} + +func (h *Handler) logError(message string, err error) { + if h.config.LogError != nil { + h.config.LogError(message, err) + } +} + +// RegisterRoutes wires GET/PUT /api/v1/system/retention. Both routes are +// admin-scoped, and GET is readable while retention is disabled +// (AC-OFFICE-RUN-HISTORY-RETENTION-004.8). +func RegisterRoutes(admin *gin.RouterGroup, handler *Handler) { + admin.GET("/retention", handler.getRetention) + admin.PUT("/retention", handler.putRetention) +} + +// Status is the GET response body: effective settings, the most recently +// completed sweep (nil before the first one — AC-004.7), the +// separately-held skip record, and the per-thresholded-table retained-count +// census (AC-004.6). +type Status struct { + Settings Settings `json:"settings"` + LastSweep *LastSweep `json:"last_sweep"` + SkipCount int64 `json:"skip_count"` + LastSkipAt *time.Time `json:"last_skip_at,omitempty"` + RetainedCounts RetainedCounts `json:"retained_counts"` +} + +func (h *Handler) getRetention(c *gin.Context) { + settings, err := h.config.SettingsStore.GetSettings(c.Request.Context()) + if err != nil { + h.logError("failed to load retention settings", err) + } + + status := Status{ + Settings: settings, + RetainedCounts: h.config.Sweeper.CensusSnapshot(), + } + if last, ok := h.config.Sweeper.LastSweepSnapshot(); ok { + status.LastSweep = &last + } + if count, lastAt := h.config.Sweeper.SkipSnapshot(); count > 0 { + status.SkipCount = count + status.LastSkipAt = &lastAt + } + c.JSON(http.StatusOK, status) +} + +func (h *Handler) putRetention(c *gin.Context) { + body, err := c.GetRawData() + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: "failed to read request body"}) + return + } + + settings, err := decodeRetentionSettings(body) + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()}) + return + } + + saved, err := h.config.SettingsStore.SaveSettings(c.Request.Context(), settings) + if err != nil { + if errors.Is(err, ErrValidation) { + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()}) + return + } + h.logError("failed to save retention settings", err) + c.JSON(http.StatusInternalServerError, gin.H{responseErrorKey: "failed to save retention settings"}) + return + } + + if h.config.OnSettingsChanged != nil { + h.config.OnSettingsChanged(saved) + } + c.JSON(http.StatusOK, saved) +} diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go new file mode 100644 index 00000000000..8a0fffd8acb --- /dev/null +++ b/apps/backend/internal/office/retention/handler_test.go @@ -0,0 +1,285 @@ +package retention + +import ( + "bytes" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/gin-gonic/gin" + + "github.com/kandev/kandev/internal/auth/authn" +) + +// newTestRetentionRouter mirrors production wiring: one admin group guarded +// by authn.RequireAdmin, since both GET and PUT /api/v1/system/retention are +// admin-scoped (unlike storage's split read/admin groups). +func newTestRetentionRouter(handler *Handler) *gin.Engine { + router := gin.New() + router.Use(func(c *gin.Context) { + authn.SetOnGin(c, authn.Identity{UserID: "admin-1", Role: authn.RoleAdmin}) + c.Next() + }) + admin := router.Group("/api/v1/system", authn.RequireAdmin()) + RegisterRoutes(admin, handler) + return router +} + +func newTestHandler(t *testing.T) (*Handler, *Sweeper) { + t.Helper() + sweeper, _ := newTestSweeper(t) + handler := NewHandler(HandlerConfig{SettingsStore: sweeper.settingsStore, Sweeper: sweeper}) + return handler, sweeper +} + +func doRequest(router *gin.Engine, method, path string, body []byte) *httptest.ResponseRecorder { + var reader *bytes.Reader + if body != nil { + reader = bytes.NewReader(body) + } else { + reader = bytes.NewReader(nil) + } + request := httptest.NewRequest(method, path, reader) + request.Header.Set("Content-Type", "application/json") + response := httptest.NewRecorder() + router.ServeHTTP(response, request) + return response +} + +func TestGetRetention_FreshInstallReturnsDefaultsAndNilLastSweep(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var status Status + if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if status.LastSweep != nil { + t.Fatalf("LastSweep = %+v, want nil before the first sweep (AC-004.7)", status.LastSweep) + } + if status.Settings != DefaultSettings() { + t.Fatalf("Settings = %+v, want defaults", status.Settings) + } + if status.SkipCount != 0 { + t.Fatalf("SkipCount = %d, want 0", status.SkipCount) + } +} + +func TestGetRetention_ReflectsLastSweepAndCensus(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + ctx := t.Context() + + sweeper.RunCensus(ctx) + sweeper.RunSweep(ctx) // preview pass; still sets LastSweep + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var status Status + if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if status.LastSweep == nil { + t.Fatal("LastSweep = nil, want non-nil after a sweep has run") + } + if status.RetainedCounts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("RetainedCounts.OfficeRoutineRuns.State = %v, want fresh", status.RetainedCounts.OfficeRoutineRuns.State) + } +} + +func TestPutRetention_OmittedFieldTakesDocumentedDefault(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var saved Settings + if err := json.Unmarshal(response.Body.Bytes(), &saved); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if saved.Enabled { + t.Fatal("Enabled = true, want false (explicitly set)") + } + want := DefaultSettings() + if saved.SweepIntervalHours != want.SweepIntervalHours { + t.Fatalf("SweepIntervalHours = %d, want the default %d (omitted field)", saved.SweepIntervalHours, want.SweepIntervalHours) + } + if saved.RoutineRuns != want.RoutineRuns { + t.Fatalf("RoutineRuns = %+v, want the default %+v (omitted field)", saved.RoutineRuns, want.RoutineRuns) + } +} + +func TestPutRetention_RepeatedIdenticalWriteIsANoOp(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body, err := json.Marshal(DefaultSettings()) + if err != nil { + t.Fatalf("marshal: %v", err) + } + + first := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + second := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if first.Code != http.StatusOK || second.Code != http.StatusOK { + t.Fatalf("status = %d, %d, want 200, 200", first.Code, second.Code) + } + if first.Body.String() != second.Body.String() { + t.Fatalf("identical writes returned different documents:\n%s\n%s", first.Body.String(), second.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_ExplicitNullIsRejectedNamingTheField(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"runs": {"window_days": null}}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "runs.window_days") { + t.Fatalf("body = %s, want it to name runs.window_days", response.Body.String()) + } + + // Nothing written: a decade-long window must survive a null-rejected PUT. + settings := DefaultSettings() + settings.Runs.WindowDays = 3650 + if _, err := sweeper.settingsStore.SaveSettings(t.Context(), settings); err != nil { + t.Fatalf("seed SaveSettings: %v", err) + } + response = doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400", response.Code) + } + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored.Runs.WindowDays != 3650 { + t.Fatalf("Runs.WindowDays = %d, want 3650 unchanged (a rejected write must not destroy configured history)", stored.Runs.WindowDays) + } +} + +func TestPutRetention_UnknownFieldIsRejectedNamingTheField(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"windw_days": 30}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "windw_days") { + t.Fatalf("body = %s, want it to name the misspelled field", response.Body.String()) + } +} + +func TestPutRetention_FractionalNumberIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"batch_limit": 100.5}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "batch_limit") { + t.Fatalf("body = %s, want it to name batch_limit", response.Body.String()) + } +} + +func TestPutRetention_WrongTypeIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": "yes"}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "enabled") { + t.Fatalf("body = %s, want it to name enabled", response.Body.String()) + } +} + +func TestPutRetention_OutOfRangeIsRejectedAndLeavesStoredSettingsUnchanged(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"batch_limit": 1}`) // below minBatchLimit=100 + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "batch_limit") { + t.Fatalf("body = %s, want it to name batch_limit", response.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_SuccessInvokesOnSettingsChanged(t *testing.T) { + gin.SetMode(gin.TestMode) + sweeper, _ := newTestSweeper(t) + + var got Settings + var called bool + handler := NewHandler(HandlerConfig{ + SettingsStore: sweeper.settingsStore, + Sweeper: sweeper, + OnSettingsChanged: func(s Settings) { + called = true + got = s + }, + }) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + if !called { + t.Fatal("OnSettingsChanged was not called") + } + if got.Enabled { + t.Fatal("OnSettingsChanged received Enabled=true, want false") + } +} diff --git a/apps/backend/internal/office/retention/health.go b/apps/backend/internal/office/retention/health.go new file mode 100644 index 00000000000..ab8bb4eab9c --- /dev/null +++ b/apps/backend/internal/office/retention/health.go @@ -0,0 +1,229 @@ +package retention + +import ( + "context" + "fmt" + "sort" + "strings" + "time" + + "github.com/kandev/kandev/internal/health" +) + +const ( + fixURL = "/settings/system/data-storage" + fixLabel = "Review retention settings" +) + +// Checker implements health.Checker for office run history retention. Every +// issue is derived fresh from live state on each Check() — LastSweep and +// RetainedCounts are already the durable, tri-state views this design +// specifies (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7, -003.11) — so +// there is no separate stored issue map to keep in sync with them. +// +// office_retention_count_failed: is a ninth issue id beyond the +// design's closed eight-id catalogue: F33 (a Build-accepted spec gap) found +// no id for a failed census evaluation. It fires only on CensusStale (a +// table that had a successful evaluation and then failed); CensusNotComputed +// is the pre-first-success state AC-003.11 requires rendering as absent +// rather than alarming, so it raises nothing on its own. +// +// office_retention_threshold:
and office_retention_disabled:
+// both answer AC-003.5/-003.7's "retained count over threshold" condition, +// split by whether retention is enabled: AC-003.7 requires the disabled case +// to additionally state that retention is disabled, and the catalogue gives +// it its own id rather than a variable message under one id. +type Checker struct { + settingsStore *SettingsStore + sweeper *Sweeper + previewMarker *PreviewMarkerStore +} + +// NewChecker wires the health checker to the package's own stores. +func NewChecker(settingsStore *SettingsStore, sweeper *Sweeper, previewMarker *PreviewMarkerStore) *Checker { + return &Checker{settingsStore: settingsStore, sweeper: sweeper, previewMarker: previewMarker} +} + +func (c *Checker) Name() string { return "Office run retention" } +func (c *Checker) Category() string { return "office" } + +func (c *Checker) Check(ctx context.Context) []health.Issue { + var issues []health.Issue + + settings, err := c.settingsStore.GetSettings(ctx) + if err != nil { + issues = append(issues, issue( + "office_retention_settings_invalid", + "Retention settings unreadable", + fmt.Sprintf("Stored retention settings could not be read; using the documented defaults. (%s)", err.Error()), + )) + } + + if _, readable := c.previewMarker.Get(ctx); !readable { + issues = append(issues, issue( + "office_retention_preview_unreadable", + "Retention preview marker unreadable", + "The retention preview marker could not be read; office_routine_runs and runs will be previewed again on the next sweep rather than deleting.", + )) + } + + if lastSweep, ok := c.sweeper.LastSweepSnapshot(); ok { + issues = append(issues, sweptTableIssues(lastSweep, settings)...) + issues = append(issues, failedTableIssues(lastSweep)...) + } + + issues = append(issues, c.censusIssues(settings)...) + + sort.Slice(issues, func(i, j int) bool { return issues[i].ID < issues[j].ID }) + return issues +} + +// sweptTableIssues covers AC-003.2 (preview pending, one combined issue +// naming every swept table with a nonzero would-delete count) and AC-003.6 +// (backlog, per swept table). +func sweptTableIssues(last LastSweep, settings Settings) []health.Issue { + var issues []health.Issue + + type sweptEntry struct { + table TableName + result SweptTableResult + window int + } + entries := []sweptEntry{ + {TableOfficeRoutineRuns, last.OfficeRoutineRuns, settings.RoutineRuns.WindowDays}, + {TableRuns, last.Runs, settings.Runs.WindowDays}, + } + + var pending []string + for _, e := range entries { + if e.result.Previewed && e.result.WouldDelete > 0 { + pending = append(pending, fmt.Sprintf("%s: %d rows under a %d-day window", e.table, e.result.WouldDelete, e.window)) + } + } + if len(pending) > 0 { + issues = append(issues, issue( + "office_retention_preview_pending", + "Retention preview pending deletion", + "Deletion begins at the next scheduled sweep: "+strings.Join(pending, "; ")+".", + )) + } + + for _, e := range entries { + if !e.result.Backlog { + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_backlog:%s", e.table), + "Retention is behind", + fmt.Sprintf("%s has more eligible rows than one sweep's batch limit; %d rows were deleted this sweep and retention remains behind.", e.table, e.result.Deleted), + )) + } + return issues +} + +// failedTableIssues covers AC-002.7/AC-004.6's per-table sweep failure. Only +// office_routine_runs and runs ever carry a nonempty Err in the current +// sweep implementation — a satellite's own delete is never independently +// batched or retried — but every reported table is checked generically so a +// future failure mode on a satellite surfaces without a code change here. +func failedTableIssues(last LastSweep) []health.Issue { + entries := []struct { + table TableName + result TableSweepResult + }{ + {TableOfficeRoutineRuns, last.OfficeRoutineRuns.TableSweepResult}, + {TableRuns, last.Runs.TableSweepResult}, + {TableRunEvents, last.RunEvents}, + {"office_run_route_attempts", last.RouteAttempts}, + {"office_run_skills", last.RunSkills}, + } + var issues []health.Issue + for _, e := range entries { + if e.result.Err == "" { + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_failed:%s", e.table), + "Retention sweep failed", + fmt.Sprintf("The last sweep failed for %s: %s", e.table, e.result.Err), + )) + } + return issues +} + +// censusIssues covers AC-001.10 (unknown status), AC-003.5/-003.7 (threshold, +// split on enabled/disabled), and F33's office_retention_count_failed. +func (c *Checker) censusIssues(settings Settings) []health.Issue { + counts := c.sweeper.CensusSnapshot() + + entries := []struct { + table TableName + census TableCensus + warnRows int + }{ + {TableOfficeRoutineRuns, counts.OfficeRoutineRuns, settings.RoutineRuns.WarnRows}, + {TableRuns, counts.Runs, settings.Runs.WarnRows}, + {TableRunEvents, counts.RunEvents, settings.RunEvents.WarnRows}, + } + + var issues []health.Issue + for _, e := range entries { + if e.census.State == CensusStale { + issues = append(issues, issue( + fmt.Sprintf("office_retention_count_failed:%s", e.table), + "Retained-row count evaluation failing", + fmt.Sprintf("%s's retained-row count could not be re-evaluated; showing the last successful count from %s.", e.table, e.census.AsOf.Format(time.RFC3339)), + )) + } + if e.census.State == CensusNotComputed { + continue + } + + if len(e.census.UnknownStatuses) > 0 { + issues = append(issues, issue( + fmt.Sprintf("office_retention_unknown_status:%s", e.table), + "Unrecognized status in retained rows", + fmt.Sprintf("%s has rows with unrecognized status values, treated as live state and never pruned: %s.", e.table, strings.Join(e.census.UnknownStatuses, ", ")), + )) + } + + if e.warnRows <= 0 || e.census.RetainedCount <= int64(e.warnRows) { + continue + } + message := thresholdMessage(e.table, e.census, e.warnRows) + if !settings.Enabled { + issues = append(issues, issue( + fmt.Sprintf("office_retention_disabled:%s", e.table), + "Retention disabled with rows over threshold", + message+" Retention is disabled, so this table is not being pruned.", + )) + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_threshold:%s", e.table), + "Retained rows over threshold", + message, + )) + } + return issues +} + +func thresholdMessage(table TableName, census TableCensus, warnRows int) string { + message := fmt.Sprintf("%s has %d retained rows, over its threshold of %d.", table, census.RetainedCount, warnRows) + if census.TopRoutineID != "" { + message += fmt.Sprintf(" Routine %s holds the largest share of retained rows, %.1f%%.", census.TopRoutineID, census.TopRoutineShare*100) + } + return message +} + +func issue(id, title, message string) health.Issue { + return health.Issue{ + ID: id, + Category: "office", + Title: title, + Message: message, + Severity: health.SeverityWarning, + FixURL: fixURL, + FixLabel: fixLabel, + } +} diff --git a/apps/backend/internal/office/retention/health_test.go b/apps/backend/internal/office/retention/health_test.go new file mode 100644 index 00000000000..22b38178ad2 --- /dev/null +++ b/apps/backend/internal/office/retention/health_test.go @@ -0,0 +1,364 @@ +package retention + +import ( + "context" + "errors" + "testing" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/health" +) + +var errCensusEvaluation = errors.New("census evaluation failed") + +func newTestChecker(t *testing.T) (*Checker, *Sweeper, *sqlx.DB) { + t.Helper() + sweeper, conn := newTestSweeper(t) + checker := NewChecker(sweeper.settingsStore, sweeper, sweeper.previewMarker) + return checker, sweeper, conn +} + +func issueIDs(issues []health.Issue) []string { + ids := make([]string, 0, len(issues)) + for _, i := range issues { + ids = append(ids, i.ID) + } + return ids +} + +func hasIssue(issues []health.Issue, id string) bool { + for _, i := range issues { + if i.ID == id { + return true + } + } + return false +} + +func TestChecker_NameAndCategory(t *testing.T) { + checker, _, _ := newTestChecker(t) + if checker.Name() != "Office run retention" { + t.Fatalf("Name() = %q", checker.Name()) + } + if checker.Category() != "office" { + t.Fatalf("Category() = %q", checker.Category()) + } +} + +func TestChecker_FreshInstallNoSweepNoIssues(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + sweeper.RunCensus(ctx) // AC-003.11: counts available before any sweep + + issues := checker.Check(ctx) + if len(issues) != 0 { + t.Fatalf("issues = %v, want none on a fresh, empty, enabled install", issueIDs(issues)) + } +} + +func TestChecker_SettingsInvalidRaisesGlobalIssue(t *testing.T) { + checker, _, conn := newTestChecker(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_settings_invalid") { + t.Fatalf("issues = %v, want office_retention_settings_invalid", issueIDs(issues)) + } +} + +func TestChecker_PreviewMarkerUnreadableRaisesGlobalIssue(t *testing.T) { + checker, _, conn := newTestChecker(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention_preview_completed', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable preview marker: %v", err) + } + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_preview_unreadable") { + t.Fatalf("issues = %v, want office_retention_preview_unreadable", issueIDs(issues)) + } +} + +func TestChecker_PreviewPendingNamesTablesWithNonzeroWouldDelete(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) // preview pass, both tables have 1 eligible row + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_preview_pending") { + t.Fatalf("issues = %v, want office_retention_preview_pending", issueIDs(issues)) + } +} + +func TestChecker_PreviewWithNothingEligibleRaisesNoWarning(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + sweeper.RunSweep(ctx) // preview pass, nothing seeded, both tables report zero + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_preview_pending") { + t.Fatalf("issues = %v, want no preview_pending when the preview found nothing (AC-003.3)", issueIDs(issues)) + } +} + +func TestChecker_BacklogRaisesPerSweptTable(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass, 105 eligible > batch limit 100 + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_backlog:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_backlog:office_routine_runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_backlog:runs") { + t.Fatalf("issues = %v, want no backlog issue for runs (never seeded)", issueIDs(issues)) + } +} + +func TestChecker_PreviewedTableNeverRaisesBacklog(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass only: 105 eligible, never backlog (AC-002.3) + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_backlog:office_routine_runs") { + t.Fatalf("issues = %v, want no backlog issue for a preview pass regardless of eligible count", issueIDs(issues)) + } +} + +func TestChecker_AbandonedRunsBatchRaisesFailedIssueForRunsOnly(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + old := daysAgo(60) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + + testBeforeSelectEligibleRunIDs = func(attempt int) { + if attempt == 0 { + return + } + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID) + } + testAfterSelectEligibleRunIDs = func(int, []string) { + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID) + } + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + + sweeper.RunSweep(ctx) // deleting pass: forced to abandon + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_failed:runs") { + t.Fatalf("issues = %v, want office_retention_failed:runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_failed:run_events") { + t.Fatalf("issues = %v, want no failed issue for run_events (the rollback restored it, not a failure of its own)", issueIDs(issues)) + } +} + +func TestChecker_UnknownStatusRaisesWhileDisabledAndBeforeAnySweep(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + + sweeper.RunCensus(ctx) // AC-003.11: census runs independent of the sweep/enabled state + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_unknown_status:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_unknown_status:office_routine_runs (001.10, disabled, no sweep ever ran)", issueIDs(issues)) + } +} + +func TestChecker_ThresholdExceededWhileEnabledRaisesThresholdID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_threshold:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_threshold:office_routine_runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want no disabled-variant issue while enabled", issueIDs(issues)) + } +} + +func TestChecker_ThresholdExceededWhileDisabledRaisesDisabledID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + settings.RoutineRuns.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_disabled:office_routine_runs (AC-003.7)", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_threshold:office_routine_runs") { + t.Fatalf("issues = %v, want no plain threshold issue while disabled", issueIDs(issues)) + } +} + +func TestChecker_ZeroWarnRowsDisablesThresholdIssue(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + for i := 0; i < 5; i++ { + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(i+1)), daysAgo(i+1)) + } + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_threshold:office_routine_runs") || hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want no threshold issue when warn_rows=0", issueIDs(issues)) + } +} + +func TestChecker_CountFailedOnlyAfterAPriorSuccess(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + sweeper.RunCensus(ctx) // succeeds once + + sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_count_failed:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_count_failed:office_routine_runs after a prior success then a failure", issueIDs(issues)) + } +} + +func TestChecker_NotComputedNeverRaisesCountFailed(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + + sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation) // fails, no prior success + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_count_failed:office_routine_runs") { + t.Fatalf("issues = %v, want no count_failed issue before any evaluation ever succeeded (AC-003.11's not-yet-computed state)", issueIDs(issues)) + } +} + +func TestChecker_IssuesSortedByID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 1 + settings.Runs.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(2)), daysAgo(2)) + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + ids := issueIDs(issues) + for i := 1; i < len(ids); i++ { + if ids[i-1] > ids[i] { + t.Fatalf("issues not sorted by id: %v", ids) + } + } +} diff --git a/apps/backend/internal/office/retention/lock.go b/apps/backend/internal/office/retention/lock.go new file mode 100644 index 00000000000..ab0febeb18e --- /dev/null +++ b/apps/backend/internal/office/retention/lock.go @@ -0,0 +1,109 @@ +package retention + +import ( + "context" + "database/sql/driver" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" +) + +// advisoryLockKey is retention's own PostgreSQL advisory lock namespace, +// distinct from every other hashtextextended(?, 0) call site in the repo +// (participants.go, secrets/sqlite_store.go, workflow/repository/phase2_sqlite.go) +// so a sweep never contends with participant-seat, secret-transfer, or +// workflow-phase locking. +const advisoryLockKey = "office_run_retention_sweep" + +const unlockTimeout = 5 * time.Second + +// sweepSession is a PostgreSQL session-scoped, non-blocking exclusivity +// lock for one sweep (AC-OFFICE-RUN-HISTORY-RETENTION-002.12). Its shape +// resolves three findings the operator accepted as Build's to decide +// (F25, F26, F27 in the task plan): +// +// - F27 (single connection budget): every statement the sweep issues — +// count, delete, census — runs through this one dedicated connection, +// returned by queryer(), rather than reserving it for the lock alone +// and running batches on the shared pool. Reserving a second +// connection is exactly what a maxOpenConns=1 pool (what +// testutil.OpenIsolatedPostgres, the mandated Postgres-gated test +// harness, sets) cannot supply; one connection total removes the +// deadlock. +// - F25 (exclusivity window): because every sweep statement runs on the +// lock connection, there is no window where work proceeds on a +// different connection after the session died — if the session ends, +// the very next statement on it fails immediately instead of +// continuing to run against the pool while another backend has +// already re-acquired the lock. The between-tables alive() check the +// design specifies is kept anyway, as a cheap early exit before +// starting a table's work rather than the only guard against loss of +// exclusivity. +// - F26 (release safety): release() unlocks and closes on an independent +// context, not the sweep's (which may already be cancelled), with a +// bounded timeout, and discards the connection via driver.ErrBadConn +// whenever the unlock did not provably succeed — so database/sql +// never pools a session that may still hold the lock, which is what +// the design's "cannot wedge retention permanently" claim actually +// requires (internal/db sets no ConnMaxLifetime). +type sweepSession struct { + conn *sqlx.Conn +} + +// acquireSweepSession tries to take the advisory lock on a fresh dedicated +// connection. ok is false when another backend already holds it or the +// connection could not be checked out; the caller records a skip and +// returns rather than retrying (AC-OFFICE-RUN-HISTORY-RETENTION-002.12). +func acquireSweepSession(ctx context.Context, pool *db.Pool) (*sweepSession, bool, error) { + conn, err := pool.Writer().Connx(ctx) + if err != nil { + return nil, false, err + } + + var acquired bool + err = conn.GetContext(ctx, &acquired, `SELECT pg_try_advisory_lock(hashtextextended($1, 0))`, advisoryLockKey) + if err != nil { + _ = conn.Close() + return nil, false, err + } + if !acquired { + _ = conn.Close() + return nil, false, nil + } + return &sweepSession{conn: conn}, true, nil +} + +// queryer is the connection every sweep statement must run through for the +// whole sweep's duration (F27/F25 above). +func (s *sweepSession) queryer() queryer { + return s.conn +} + +// alive reports whether the lock connection is still usable. Checked +// between tables; the sweep stops before the next table when this returns +// false and does not attempt to re-acquire, since a re-acquisition after +// another backend has taken the lock would produce exactly the concurrent +// sweep AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent. +func (s *sweepSession) alive(ctx context.Context) bool { + return s.conn.PingContext(ctx) == nil +} + +// release unlocks and closes the session. Always safe to call once; never +// call it twice. +func (s *sweepSession) release() { + ctx, cancel := context.WithTimeout(context.Background(), unlockTimeout) + defer cancel() + + var unlocked bool + err := s.conn.GetContext(ctx, &unlocked, `SELECT pg_advisory_unlock(hashtextextended($1, 0))`, advisoryLockKey) + if err != nil || !unlocked { + // The unlock did not provably succeed: force database/sql to + // discard this connection instead of returning it to the pool, + // so a session that may still hold the lock can never be reused + // by a later, unrelated caller (F26). + _ = s.conn.Raw(func(driverConn any) error { return driver.ErrBadConn }) + } + _ = s.conn.Close() +} diff --git a/apps/backend/internal/office/retention/lock_postgres_test.go b/apps/backend/internal/office/retention/lock_postgres_test.go new file mode 100644 index 00000000000..8912887e0aa --- /dev/null +++ b/apps/backend/internal/office/retention/lock_postgres_test.go @@ -0,0 +1,141 @@ +package retention + +import ( + "context" + "testing" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/testutil" +) + +// TestSweepSession_SecondBackendSkipsRatherThanBlocks proves +// AC-OFFICE-RUN-HISTORY-RETENTION-002.12: a second backend racing for the +// same advisory lock gets ok=false immediately rather than waiting. +func TestSweepSession_SecondBackendSkipsRatherThanBlocks(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + first := testutil.OpenIsolatedPostgres(t, dsn) + firstPool := db.NewPool(first, first) + winner, ok, err := acquireSweepSession(ctx, firstPool) + if err != nil { + t.Fatalf("acquireSweepSession (winner): %v", err) + } + if !ok { + t.Fatal("winner: ok = false, want true") + } + defer winner.release() + + second := testutil.OpenIsolatedPostgres(t, dsn) + secondPool := db.NewPool(second, second) + loser, ok, err := acquireSweepSession(ctx, secondPool) + if err != nil { + t.Fatalf("acquireSweepSession (loser): %v", err) + } + if ok { + loser.release() + t.Fatal("loser: ok = true, want false") + } +} + +// TestSweepSession_RunsOnMaxOpenConnsOnePool is F27's regression test: the +// mandated Postgres test harness (testutil.OpenIsolatedPostgres) opens with +// SetMaxOpenConns(1). Reserving the lock connection AND running batches on +// the shared pool would deadlock there, since no second connection is ever +// available. Routing every statement through the lock connection's own +// queryer() must not deadlock and must give a winner more than one table's +// worth of work to do, so a transaction-scoped lock would have incorrectly +// looked sufficient here. +func TestSweepSession_RunsOnMaxOpenConnsOnePool(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + conn := testutil.OpenIsolatedPostgres(t, dsn) + pool := db.NewPool(conn, conn) + + session, ok, err := acquireSweepSession(ctx, pool) + if err != nil { + t.Fatalf("acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("ok = false, want true") + } + defer session.release() + + q := session.queryer() + for i := 0; i < 3; i++ { + var one int + if err := q.GetContext(ctx, &one, `SELECT 1`); err != nil { + t.Fatalf("query %d on lock connection: %v", i, err) + } + if one != 1 { + t.Fatalf("query %d = %d, want 1", i, one) + } + } +} + +// TestSweepSession_ReleaseAllowsReacquisition proves release() actually +// drops the lock rather than merely closing a connection database/sql +// might still consider live for pooling purposes. +func TestSweepSession_ReleaseAllowsReacquisition(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + conn := testutil.OpenIsolatedPostgres(t, dsn) + pool := db.NewPool(conn, conn) + + first, ok, err := acquireSweepSession(ctx, pool) + if err != nil || !ok { + t.Fatalf("first acquire: ok=%v err=%v", ok, err) + } + first.release() + + second, ok, err := acquireSweepSession(ctx, pool) + if err != nil { + t.Fatalf("second acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("second acquire: ok = false, want true after release") + } + second.release() +} + +// TestSweepSession_AliveFalseAfterSessionTerminated proves the +// between-tables liveness check (F25's early exit) actually detects a +// dropped session, and that PostgreSQL's own advisory-lock self-healing +// (F26's justification for a session lock over a lease row) lets a new +// session acquire afterward. +func TestSweepSession_AliveFalseAfterSessionTerminated(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + admin := testutil.OpenIsolatedPostgres(t, dsn) + adminPool := db.NewPool(admin, admin) + + victimConn := testutil.OpenIsolatedPostgres(t, dsn) + victimPool := db.NewPool(victimConn, victimConn) + victim, ok, err := acquireSweepSession(ctx, victimPool) + if err != nil || !ok { + t.Fatalf("victim acquire: ok=%v err=%v", ok, err) + } + + var pid int + if err := victim.conn.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil { + t.Fatalf("select pg_backend_pid: %v", err) + } + if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil { + t.Fatalf("terminate victim backend: %v", err) + } + + if victim.alive(ctx) { + t.Fatal("alive() = true after backend termination, want false") + } + + replacement, ok, err := acquireSweepSession(ctx, adminPool) + if err != nil { + t.Fatalf("replacement acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("replacement: ok = false, want true (terminated session must release the lock)") + } + replacement.release() +} diff --git a/apps/backend/internal/office/retention/metrics_vars.go b/apps/backend/internal/office/retention/metrics_vars.go new file mode 100644 index 00000000000..dc12c9cd591 --- /dev/null +++ b/apps/backend/internal/office/retention/metrics_vars.go @@ -0,0 +1,55 @@ +package retention + +import ( + "expvar" + "strings" +) + +// expvar maps published at package init, exposed via stdlib's /debug/vars +// handler, mirroring internal/office/scheduler/metrics_vars.go's label +// model. Development convenience only (see the design's expvar +// section) — nothing in REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on +// these; the durable operator surfaces are structured logs and +// health.Issue. +var ( + retentionSweepTotal = expvar.NewMap("office_retention_sweep_total") + retentionDeletedTotal = expvar.NewMap("office_retention_deleted_total") + retentionCensusTotal = expvar.NewMap("office_retention_census_total") +) + +// metricLabel builds a "k1=v1;k2=v2;..." label string for an expvar map +// key, matching office/scheduler's format so a downstream parser handles +// both packages identically. +func metricLabel(pairs ...string) string { + if len(pairs)%2 != 0 { + return "" + } + parts := make([]string, 0, len(pairs)/2) + for i := 0; i < len(pairs); i += 2 { + parts = append(parts, pairs[i]+"="+pairs[i+1]) + } + return strings.Join(parts, ";") +} + +func incSweepCompleted() { + retentionSweepTotal.Add(metricLabel("outcome", "completed"), 1) +} + +func incSweepSkipped() { + retentionSweepTotal.Add(metricLabel("outcome", "skipped"), 1) +} + +func incDeleted(table TableName, n int64) { + if n <= 0 { + return + } + retentionDeletedTotal.Add(metricLabel("table", string(table)), n) +} + +func incCensus(table TableName, err error) { + outcome := "fresh" + if err != nil { + outcome = "stale" + } + retentionCensusTotal.Add(metricLabel("table", string(table), "outcome", outcome), 1) +} diff --git a/apps/backend/internal/office/retention/metrics_vars_test.go b/apps/backend/internal/office/retention/metrics_vars_test.go new file mode 100644 index 00000000000..d9b25a1856d --- /dev/null +++ b/apps/backend/internal/office/retention/metrics_vars_test.go @@ -0,0 +1,116 @@ +package retention + +import ( + "errors" + "expvar" + "strconv" + "strings" + "testing" +) + +// readCounter walks the expvar map looking for a key that matches the +// supplied prefix. Returns 0 when no key matches. The prefix match keeps +// the assertion robust against process-wide test pollution. +func readCounter(t *testing.T, m *expvar.Map, prefix string) int64 { + t.Helper() + var total int64 + m.Do(func(kv expvar.KeyValue) { + if !strings.HasPrefix(kv.Key, prefix) { + return + } + n, err := strconv.ParseInt(kv.Value.String(), 10, 64) + if err != nil { + t.Fatalf("counter %q value not int: %s", kv.Key, kv.Value.String()) + } + total += n + }) + return total +} + +func TestMetricLabel(t *testing.T) { + cases := []struct { + name string + pairs []string + want string + }{ + {"single_pair", []string{"table", "runs"}, "table=runs"}, + {"odd_args_returns_empty", []string{"table"}, ""}, + {"two_pairs", []string{"table", "runs", "outcome", "completed"}, "table=runs;outcome=completed"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := metricLabel(tc.pairs...); got != tc.want { + t.Errorf("metricLabel(%v) = %q, want %q", tc.pairs, got, tc.want) + } + }) + } +} + +func TestIncSweepCompletedAndSkipped(t *testing.T) { + beforeCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed")) + incSweepCompleted() + afterCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed")) + if afterCompleted-beforeCompleted != 1 { + t.Errorf("sweep completed delta = %d, want 1", afterCompleted-beforeCompleted) + } + + beforeSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped")) + incSweepSkipped() + afterSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped")) + if afterSkipped-beforeSkipped != 1 { + t.Errorf("sweep skipped delta = %d, want 1", afterSkipped-beforeSkipped) + } +} + +func TestIncDeleted_ZeroIsNotRecorded(t *testing.T) { + label := metricLabel("table", "test_table_zero") + before := readCounter(t, retentionDeletedTotal, label) + incDeleted("test_table_zero", 0) + after := readCounter(t, retentionDeletedTotal, label) + if after != before { + t.Errorf("delta = %d, want 0 (a zero delete must not create a counter entry)", after-before) + } +} + +func TestIncDeleted_PositiveIsRecorded(t *testing.T) { + label := metricLabel("table", "test_table_positive") + before := readCounter(t, retentionDeletedTotal, label) + incDeleted("test_table_positive", 7) + after := readCounter(t, retentionDeletedTotal, label) + if after-before != 7 { + t.Errorf("delta = %d, want 7", after-before) + } +} + +func TestIncCensus_NilErrorIsFresh(t *testing.T) { + label := metricLabel("table", "test_census_fresh", "outcome", "fresh") + before := readCounter(t, retentionCensusTotal, label) + incCensus("test_census_fresh", nil) + after := readCounter(t, retentionCensusTotal, label) + if after-before != 1 { + t.Errorf("fresh delta = %d, want 1", after-before) + } +} + +func TestIncCensus_ErrorIsStale(t *testing.T) { + label := metricLabel("table", "test_census_stale", "outcome", "stale") + before := readCounter(t, retentionCensusTotal, label) + incCensus("test_census_stale", errors.New("boom")) + after := readCounter(t, retentionCensusTotal, label) + if after-before != 1 { + t.Errorf("stale delta = %d, want 1", after-before) + } +} + +func TestExpvarMapsPublishedAtKnownNames(t *testing.T) { + expected := []string{ + "office_retention_sweep_total", + "office_retention_deleted_total", + "office_retention_census_total", + } + for _, name := range expected { + if expvar.Get(name) == nil { + t.Errorf("expvar %q not published — /debug/vars consumers will miss it", name) + } + } +} diff --git a/apps/backend/internal/office/retention/policy.go b/apps/backend/internal/office/retention/policy.go new file mode 100644 index 00000000000..b3b7b4b6370 --- /dev/null +++ b/apps/backend/internal/office/retention/policy.go @@ -0,0 +1,59 @@ +package retention + +// StatusClass distinguishes a row eligible for age-based deletion from a row +// a live decision still reads, and from a row whose status this package does +// not recognize at all — a status classification failure, not a bare bool, +// so an unrecognized status can be told apart from a recognized live-state +// one (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +type StatusClass int + +const ( + // StatusUnknown is a status belonging to neither a table's history set + // nor its live-state set. Treated as live state and warned about. + StatusUnknown StatusClass = iota + StatusHistory + StatusLiveState +) + +// RoutineRunHistoryStatuses are the office_routine_runs statuses eligible +// for age-based deletion (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). +var RoutineRunHistoryStatuses = []string{"skipped", "coalesced", "failed", "done", "cancelled"} + +// RoutineRunLiveStatuses are the office_routine_runs statuses that are +// never age-pruned, at any age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). +var RoutineRunLiveStatuses = []string{"received", "task_created"} + +// RunHistoryStatuses are the runs statuses eligible for age-based deletion +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). cancelled is history: its only +// writer, CancelRunsWhere, moves a row there from queued/claimed and stamps +// finished_at in the same statement. +var RunHistoryStatuses = []string{"finished", "failed", "cancelled"} + +// RunLiveStatuses are the runs statuses that are never age-pruned, at any +// age, including a run parked for a future routing retry +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). +var RunLiveStatuses = []string{"queued", "claimed"} + +// ClassifyRoutineRunStatus classifies an office_routine_runs.status value. +func ClassifyRoutineRunStatus(status string) StatusClass { + return classify(status, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) +} + +// ClassifyRunStatus classifies a runs.status value. +func ClassifyRunStatus(status string) StatusClass { + return classify(status, RunHistoryStatuses, RunLiveStatuses) +} + +func classify(status string, history, live []string) StatusClass { + for _, s := range history { + if s == status { + return StatusHistory + } + } + for _, s := range live { + if s == status { + return StatusLiveState + } + } + return StatusUnknown +} diff --git a/apps/backend/internal/office/retention/policy_test.go b/apps/backend/internal/office/retention/policy_test.go new file mode 100644 index 00000000000..d4626a1fc20 --- /dev/null +++ b/apps/backend/internal/office/retention/policy_test.go @@ -0,0 +1,98 @@ +package retention + +import ( + "sort" + "testing" + + "github.com/kandev/kandev/internal/office/models" +) + +func TestClassifyRoutineRunStatus(t *testing.T) { + for _, s := range RoutineRunHistoryStatuses { + if got := ClassifyRoutineRunStatus(s); got != StatusHistory { + t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusHistory", s, got) + } + } + for _, s := range RoutineRunLiveStatuses { + if got := ClassifyRoutineRunStatus(s); got != StatusLiveState { + t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusLiveState", s, got) + } + } + if got := ClassifyRoutineRunStatus("some_future_status"); got != StatusUnknown { + t.Errorf("ClassifyRoutineRunStatus(unrecognized) = %v, want StatusUnknown", got) + } +} + +func TestClassifyRunStatus(t *testing.T) { + for _, s := range RunHistoryStatuses { + if got := ClassifyRunStatus(s); got != StatusHistory { + t.Errorf("ClassifyRunStatus(%q) = %v, want StatusHistory", s, got) + } + } + for _, s := range RunLiveStatuses { + if got := ClassifyRunStatus(s); got != StatusLiveState { + t.Errorf("ClassifyRunStatus(%q) = %v, want StatusLiveState", s, got) + } + } + if got := ClassifyRunStatus("some_future_status"); got != StatusUnknown { + t.Errorf("ClassifyRunStatus(unrecognized) = %v, want StatusUnknown", got) + } +} + +// TestRoutineRunStatusSets_CoverEveryEnumValue pins the closed-set claim in +// the system design: history + live-state must equal exactly the +// office/models RoutineRunStatus enumeration. A future eighth status added +// to enums.go without updating this file fails here rather than silently +// falling through ClassifyRoutineRunStatus's StatusUnknown branch on every +// install that never runs this test — the exact gap that hid `cancelled`. +func TestRoutineRunStatusSets_CoverEveryEnumValue(t *testing.T) { + wantAll := []string{ + models.RoutineRunStatusReceived.String(), + models.RoutineRunStatusTaskCreated.String(), + models.RoutineRunStatusSkipped.String(), + models.RoutineRunStatusCoalesced.String(), + models.RoutineRunStatusFailed.String(), + models.RoutineRunStatusDone.String(), + models.RoutineRunStatusCancelled.String(), + } + gotAll := append(append([]string{}, RoutineRunHistoryStatuses...), RoutineRunLiveStatuses...) + assertSameSet(t, "office_routine_runs", wantAll, gotAll) +} + +// TestRunStatusSets_CoverEveryLiteralSQLWriter pins the closed-set claim for +// runs.status. models.RunStatus itself is incomplete (it does not list +// "cancelled", even though CancelRunsWhere in +// internal/runs/repository/sqlite/cancel.go writes it) — this test asserts +// against the actual literal statuses written by SQL in that package +// instead, per the system design's Testing section, so the enum's own +// incompleteness cannot mask a gap here the way it masked `cancelled` +// before this capability existed. +func TestRunStatusSets_CoverEveryLiteralSQLWriter(t *testing.T) { + // Literal statuses written by internal/runs/repository/sqlite: + // cancel.go: 'cancelled' (CancelRunsWhere) + // runs.go:172: 'claimed' (ClaimNextRun family) + // runs.go:512: 'claimed' (claim by reason) + // runs.go:539: 'queued' (ScheduleRetry) + // runs.go:562: 'queued' (recoverStaleClaimed) + // runs.go:195/737: status passed as a bound parameter, produced by + // callers with 'finished' or 'failed' (FinishRun/MarkRunFailed). + wantAll := []string{"queued", "claimed", "finished", "failed", "cancelled"} + gotAll := append(append([]string{}, RunHistoryStatuses...), RunLiveStatuses...) + assertSameSet(t, "runs", wantAll, gotAll) +} + +func assertSameSet(t *testing.T, table string, want, got []string) { + t.Helper() + w := append([]string{}, want...) + g := append([]string{}, got...) + sort.Strings(w) + sort.Strings(g) + if len(w) != len(g) { + t.Fatalf("%s: status set size = %d, want %d (got=%v want=%v)", table, len(g), len(w), g, w) + } + for i := range w { + if w[i] != g[i] { + t.Fatalf("%s: status set mismatch at %d: got %v, want %v", table, i, g, w) + } + } +} diff --git a/apps/backend/internal/office/retention/preview_marker.go b/apps/backend/internal/office/retention/preview_marker.go new file mode 100644 index 00000000000..dc668b8ef21 --- /dev/null +++ b/apps/backend/internal/office/retention/preview_marker.go @@ -0,0 +1,117 @@ +package retention + +import ( + "context" + "database/sql" + "encoding/json" + "errors" + "time" + + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// previewMarkerKey is the settings-store key holding the per-swept-table +// preview completion marker, a JSON object keyed by table name. +const previewMarkerKey = "office_run_retention_preview_completed" + +// PreviewMarker records, per swept table, the timestamp its preview +// completed. A table absent from the map has never completed a preview. +type PreviewMarker map[TableName]time.Time + +// PreviewMarkerStore persists and reads the preview marker. +type PreviewMarkerStore struct { + store *systemsettings.Store +} + +// NewPreviewMarkerStore wraps the shared key/value settings store. +func NewPreviewMarkerStore(store *systemsettings.Store) *PreviewMarkerStore { + return &PreviewMarkerStore{store: store} +} + +// Get reads the marker. readable is false only when a document is present +// but cannot be read or parsed (AC-OFFICE-RUN-HISTORY-RETENTION-003.10); a +// document that was never written is the ordinary fresh-install state and +// reports readable=true with an empty marker, not an error. Either way, a +// table absent from the returned marker has not completed a preview. +func (s *PreviewMarkerStore) Get(ctx context.Context) (PreviewMarker, bool) { + raw, found, err := s.store.Get(ctx, previewMarkerKey) + if err != nil { + return PreviewMarker{}, false + } + if !found { + return PreviewMarker{}, true + } + var doc PreviewMarker + if err := json.Unmarshal(raw, &doc); err != nil { + return PreviewMarker{}, false + } + if doc == nil { + doc = PreviewMarker{} + } + return doc, true +} + +// MarkCompleted records that table's preview as completed at the given +// time. It is only ever called after that table's preview evaluation +// completed successfully, is never cleared by a settings change or +// restart, and never touches another table's entry. If the stored document +// was corrupt, this write replaces it with a fresh document carrying only +// this table's entry — the safe direction, since a spurious re-preview of +// another table costs one sweep and deletes nothing. +func (s *PreviewMarkerStore) MarkCompleted(ctx context.Context, table TableName, at time.Time) error { + marker, readable := s.Get(ctx) + if !readable { + marker = PreviewMarker{} + } + marker[table] = at + raw, err := json.Marshal(marker) + if err != nil { + return err + } + return s.store.Save(ctx, previewMarkerKey, raw) +} + +// GetWith is Get against an explicit connection instead of the shared +// settings pool. Required mid-sweep on PostgreSQL: every statement in a +// sweep must run on the session holding the advisory lock (F27 — see +// sweep.go and lock.go), and going through the pool here would request a +// second connection, which deadlocks under a maxOpenConns=1 pool (the +// mandated Postgres-gated test harness). +func (s *PreviewMarkerStore) GetWith(ctx context.Context, q queryer) (PreviewMarker, bool) { + var raw string + err := q.GetContext(ctx, &raw, q.Rebind(`SELECT value FROM settings WHERE key = ?`), previewMarkerKey) + if err != nil { + if errors.Is(err, sql.ErrNoRows) { + return PreviewMarker{}, true + } + return PreviewMarker{}, false + } + var doc PreviewMarker + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return PreviewMarker{}, false + } + if doc == nil { + doc = PreviewMarker{} + } + return doc, true +} + +// MarkCompletedWith is MarkCompleted against an explicit connection; see +// GetWith. +func (s *PreviewMarkerStore) MarkCompletedWith(ctx context.Context, q queryer, table TableName, at time.Time) error { + marker, readable := s.GetWith(ctx, q) + if !readable { + marker = PreviewMarker{} + } + marker[table] = at + raw, err := json.Marshal(marker) + if err != nil { + return err + } + _, err = q.ExecContext(ctx, q.Rebind(` + INSERT INTO settings (key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at + `), previewMarkerKey, string(raw), time.Now().UTC()) + return err +} diff --git a/apps/backend/internal/office/retention/preview_marker_test.go b/apps/backend/internal/office/retention/preview_marker_test.go new file mode 100644 index 00000000000..8080eb380dc --- /dev/null +++ b/apps/backend/internal/office/retention/preview_marker_test.go @@ -0,0 +1,117 @@ +package retention + +import ( + "context" + "testing" + "time" +) + +func TestPreviewMarkerStore_MissingIsReadableAndEmpty(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + + marker, readable := store.Get(context.Background()) + if !readable { + t.Fatalf("Get() readable = false, want true for a never-written marker") + } + if len(marker) != 0 { + t.Fatalf("Get() marker = %+v, want empty", marker) + } +} + +func TestPreviewMarkerStore_MarkCompletedThenGetRoundTrips(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + ctx := context.Background() + + at := time.Date(2026, 9, 9, 12, 0, 0, 0, time.UTC) + if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, at); err != nil { + t.Fatalf("MarkCompleted: %v", err) + } + + marker, readable := store.Get(ctx) + if !readable { + t.Fatalf("Get() readable = false after a valid write") + } + got, ok := marker[TableOfficeRoutineRuns] + if !ok { + t.Fatalf("marker missing office_routine_runs entry: %+v", marker) + } + if !got.Equal(at) { + t.Fatalf("marker[office_routine_runs] = %v, want %v", got, at) + } + if _, ok := marker[TableRuns]; ok { + t.Fatalf("marker has an entry for runs before it was ever marked: %+v", marker) + } +} + +// TestPreviewMarkerStore_PerTableNotGlobal proves the marker is per swept +// table, not per database: marking one table previewed must not mark a +// sibling table previewed too (AC-OFFICE-RUN-HISTORY-RETENTION-003.4). +func TestPreviewMarkerStore_PerTableNotGlobal(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + ctx := context.Background() + + if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, time.Now().UTC()); err != nil { + t.Fatalf("MarkCompleted(office_routine_runs): %v", err) + } + + marker, _ := store.Get(ctx) + if _, ok := marker[TableRuns]; ok { + t.Fatalf("marking office_routine_runs previewed also marked runs: %+v", marker) + } + + if err := store.MarkCompleted(ctx, TableRuns, time.Now().UTC()); err != nil { + t.Fatalf("MarkCompleted(runs): %v", err) + } + marker, _ = store.Get(ctx) + if len(marker) != 2 { + t.Fatalf("marker after both tables previewed = %+v, want 2 entries", marker) + } +} + +// TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed proves +// AC-OFFICE-RUN-HISTORY-RETENTION-003.10: a present-but-corrupt marker +// document is treated as "not yet previewed" for every swept table, which +// is the safe direction because a spurious re-preview deletes nothing. +func TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed(t *testing.T) { + _, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable marker: %v", err) + } + + store := NewPreviewMarkerStore(raw) + marker, readable := store.Get(ctx) + if readable { + t.Fatalf("Get() readable = true for an unparseable marker, want false") + } + if len(marker) != 0 { + t.Fatalf("Get() marker = %+v on unparseable document, want empty (not-previewed)", marker) + } +} + +// TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable proves that +// marking a table previewed after a corrupt document is detected replaces +// the corrupt document rather than erroring forever. +func TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable(t *testing.T) { + _, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable marker: %v", err) + } + store := NewPreviewMarkerStore(raw) + + at := time.Now().UTC() + if err := store.MarkCompleted(ctx, TableRuns, at); err != nil { + t.Fatalf("MarkCompleted after corrupt document: %v", err) + } + marker, readable := store.Get(ctx) + if !readable { + t.Fatalf("Get() readable = false after a fresh valid write") + } + if _, ok := marker[TableRuns]; !ok { + t.Fatalf("marker missing runs entry after recovery write: %+v", marker) + } +} diff --git a/apps/backend/internal/office/retention/runtime.go b/apps/backend/internal/office/retention/runtime.go new file mode 100644 index 00000000000..06af274b44a --- /dev/null +++ b/apps/backend/internal/office/retention/runtime.go @@ -0,0 +1,61 @@ +package retention + +import ( + "context" + + "github.com/kandev/kandev/internal/db" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// Runtime composes every retention component behind one Start/Stop pair, +// mirroring internal/system/storage.Runtime's shape: a scheduler goroutine, +// an HTTP handler, and a health checker, wired together once at boot over +// the shared pool and the shared key/value settings store. +type Runtime struct { + Store *Store + SettingsStore *SettingsStore + PreviewMarker *PreviewMarkerStore + Sweeper *Sweeper + Scheduler *Scheduler + Checker *Checker + Handler *Handler +} + +// NewRuntime constructs every retention component. logError, when set, is +// forwarded to Handler for a failed settings read or write; it never +// affects the sweep, which fails closed on its own terms (see +// SettingsStore.GetSettingsForSweep). +func NewRuntime(pool *db.Pool, settingsStore *systemsettings.Store, logError func(string, error)) *Runtime { + store := NewStore(pool) + retentionSettings := NewSettingsStore(settingsStore) + previewMarker := NewPreviewMarkerStore(settingsStore) + sweeper := NewSweeper(pool, store, retentionSettings, previewMarker) + scheduler := NewScheduler(retentionSettings, sweeper, SchedulerOptions{}) + checker := NewChecker(retentionSettings, sweeper, previewMarker) + handler := NewHandler(HandlerConfig{ + SettingsStore: retentionSettings, + Sweeper: sweeper, + OnSettingsChanged: scheduler.ApplySettings, + LogError: logError, + }) + return &Runtime{ + Store: store, + SettingsStore: retentionSettings, + PreviewMarker: previewMarker, + Sweeper: sweeper, + Scheduler: scheduler, + Checker: checker, + Handler: handler, + } +} + +// Start begins the scheduler loop (census immediately, sweep gated by +// enablement — see scheduler.go). Idempotent: safe to call once at boot. +func (r *Runtime) Start(ctx context.Context) error { + return r.Scheduler.Start(ctx) +} + +// Stop cancels and joins the scheduler loop. +func (r *Runtime) Stop() { + r.Scheduler.Stop() +} diff --git a/apps/backend/internal/office/retention/runtime_test.go b/apps/backend/internal/office/retention/runtime_test.go new file mode 100644 index 00000000000..684097e4a5c --- /dev/null +++ b/apps/backend/internal/office/retention/runtime_test.go @@ -0,0 +1,79 @@ +package retention + +import ( + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +func newTestRuntime(t *testing.T) *Runtime { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsStore, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + return NewRuntime(pool, settingsStore, nil) +} + +func TestNewRuntime_WiresEveryComponentNonNil(t *testing.T) { + runtime := newTestRuntime(t) + if runtime.Store == nil || runtime.SettingsStore == nil || runtime.PreviewMarker == nil || + runtime.Sweeper == nil || runtime.Scheduler == nil || runtime.Checker == nil || runtime.Handler == nil { + t.Fatalf("Runtime has a nil component: %+v", runtime) + } +} + +func TestRuntime_StartStopIsClean(t *testing.T) { + runtime := newTestRuntime(t) + ctx := t.Context() + + if err := runtime.Start(ctx); err != nil { + t.Fatalf("Start: %v", err) + } + defer runtime.Stop() + + if err := runtime.Start(ctx); err != nil { + t.Fatalf("second Start: %v", err) + } + runtime.Stop() + runtime.Stop() // idempotent +} + +func TestRuntime_HandlerOnSettingsChangedReachesScheduler(t *testing.T) { + runtime := newTestRuntime(t) + ctx := t.Context() + if err := runtime.Start(ctx); err != nil { + t.Fatalf("Start: %v", err) + } + defer runtime.Stop() + + settings := DefaultSettings() + settings.SweepIntervalHours = 2 + saved, err := runtime.SettingsStore.SaveSettings(ctx, settings) + if err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + // Handler's OnSettingsChanged is wired to Scheduler.ApplySettings; call + // it exactly as putRetention would and confirm the running scheduler + // picked up the change (AC-004.5, without a restart). + runtime.Handler.config.OnSettingsChanged(saved) + if got := runtime.Scheduler.latestSettings(); got.SweepIntervalHours != 2 { + t.Fatalf("Scheduler.latestSettings().SweepIntervalHours = %d, want 2", got.SweepIntervalHours) + } +} diff --git a/apps/backend/internal/office/retention/scheduler.go b/apps/backend/internal/office/retention/scheduler.go new file mode 100644 index 00000000000..82811e83546 --- /dev/null +++ b/apps/backend/internal/office/retention/scheduler.go @@ -0,0 +1,174 @@ +package retention + +import ( + "context" + "sync" + "time" +) + +// firstSweepDelay is the fixed delay before the first sweep after Start, +// and after retention transitions from disabled to enabled +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.10, -002.13). Arming at the full +// interval instead — as the census timer does — would mean an install +// restarted more often than the interval never sweeps at all. +const firstSweepDelay = 5 * time.Minute + +// SchedulerOptions configures Scheduler construction. After is injectable +// for deterministic tests; production leaves it nil and gets time.After. +type SchedulerOptions struct { + After func(time.Duration) <-chan time.Time +} + +// Scheduler owns two independent timers on one goroutine, modelled on +// internal/system/storage.Scheduler: +// +// - The census timer always runs, on the configured sweep interval, +// whether or not retention is enabled (AC-OFFICE-RUN-HISTORY-RETENTION-003.11): +// a disabled install is exactly the one whose tables grow unattended, +// and the only one where an unrecognized status would otherwise never +// be noticed. +// - The sweep timer only runs while enabled, armed at firstSweepDelay +// after Start or after a disabled-to-enabled transition, and at the +// full interval thereafter (AC-OFFICE-RUN-HISTORY-RETENTION-002.10, +// -002.13). +// +// A settings change wakes the loop and re-arms both timers from the +// moment of the change (fixed-delay, not fixed-rate): the sweep timer at +// firstSweepDelay only when retention just turned on, otherwise at the +// (possibly new) full interval; the census timer always at the full +// interval. +type Scheduler struct { + settingsStore *SettingsStore + sweeper *Sweeper + after func(time.Duration) <-chan time.Time + + lifecycleMu sync.Mutex + mu sync.Mutex + cancel context.CancelFunc + wake chan struct{} + latest Settings + wg sync.WaitGroup +} + +// NewScheduler wires the scheduler to its dependencies. +func NewScheduler(settingsStore *SettingsStore, sweeper *Sweeper, options SchedulerOptions) *Scheduler { + after := options.After + if after == nil { + after = time.After + } + return &Scheduler{settingsStore: settingsStore, sweeper: sweeper, after: after} +} + +// Start begins the scheduler loop. A no-op when already running. +func (s *Scheduler) Start(ctx context.Context) error { + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + s.mu.Lock() + running := s.cancel != nil + s.mu.Unlock() + if running { + return nil + } + + settings, err := s.settingsStore.GetSettings(ctx) + if err != nil { + return err + } + + s.mu.Lock() + workerCtx, cancel := context.WithCancel(ctx) + s.cancel = cancel + s.wake = make(chan struct{}, 1) + s.latest = settings + s.wg.Add(1) + wake := s.wake + s.mu.Unlock() + + go s.run(workerCtx, settings, wake) + return nil +} + +// ApplySettings notifies a running scheduler that settings changed, +// re-arming both timers from this moment. A no-op when not running. +func (s *Scheduler) ApplySettings(settings Settings) { + s.mu.Lock() + if s.cancel == nil || s.wake == nil { + s.mu.Unlock() + return + } + s.latest = settings + wake := s.wake + s.mu.Unlock() + select { + case wake <- struct{}{}: + default: + } +} + +// Stop cancels the scheduler loop and joins it. A no-op when not running. +func (s *Scheduler) Stop() { + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + s.mu.Lock() + cancel := s.cancel + s.cancel = nil + s.wake = nil + s.mu.Unlock() + if cancel != nil { + cancel() + s.wg.Wait() + } +} + +func (s *Scheduler) run(ctx context.Context, settings Settings, wake <-chan struct{}) { + defer s.wg.Done() + + // The first census evaluation runs here, on the scheduler goroutine, + // not blocking Start's caller (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's + // "runs at Start, not at the first sweep"). + s.sweeper.RunCensus(ctx) + census := s.after(sweepInterval(settings)) + + var sweep <-chan time.Time + if settings.Enabled { + sweep = s.after(firstSweepDelay) + } + + for { + select { + case <-ctx.Done(): + return + case <-wake: + wasEnabled := settings.Enabled + settings = s.latestSettings() + + sweep = nil + if settings.Enabled { + if wasEnabled { + sweep = s.after(sweepInterval(settings)) + } else { + sweep = s.after(firstSweepDelay) + } + } + census = s.after(sweepInterval(settings)) + + case <-census: + s.sweeper.RunCensus(ctx) + census = s.after(sweepInterval(settings)) + + case <-sweep: + s.sweeper.RunSweep(ctx) + sweep = s.after(sweepInterval(settings)) + } + } +} + +func (s *Scheduler) latestSettings() Settings { + s.mu.Lock() + defer s.mu.Unlock() + return s.latest +} + +func sweepInterval(settings Settings) time.Duration { + return time.Duration(settings.SweepIntervalHours) * time.Hour +} diff --git a/apps/backend/internal/office/retention/scheduler_test.go b/apps/backend/internal/office/retention/scheduler_test.go new file mode 100644 index 00000000000..0f4406fb63e --- /dev/null +++ b/apps/backend/internal/office/retention/scheduler_test.go @@ -0,0 +1,272 @@ +package retention + +import ( + "context" + "sync" + "testing" + "time" +) + +// fakeAfter is a deterministic stand-in for time.After, keyed by duration: +// every call for the same duration returns the same channel, so a test can +// fire a specific timer (e.g. firstSweepDelay vs. the configured interval) +// without racing real wall-clock time. armed records every duration the +// scheduler has requested, in order, so a test can prove which timer was +// armed and when — including proving RunCensus already completed +// synchronously before the following after() call. +type fakeAfter struct { + mu sync.Mutex + chans map[time.Duration]chan time.Time + armed chan time.Duration +} + +func newFakeAfter() *fakeAfter { + return &fakeAfter{chans: map[time.Duration]chan time.Time{}, armed: make(chan time.Duration, 64)} +} + +func (f *fakeAfter) after(d time.Duration) <-chan time.Time { + f.mu.Lock() + ch, ok := f.chans[d] + if !ok { + ch = make(chan time.Time, 1) + f.chans[d] = ch + } + f.mu.Unlock() + f.armed <- d + return ch +} + +func (f *fakeAfter) fire(t *testing.T, d time.Duration) { + t.Helper() + f.mu.Lock() + ch, ok := f.chans[d] + f.mu.Unlock() + if !ok { + t.Fatalf("fire: no timer ever armed for duration %v", d) + } + ch <- time.Now() +} + +// waitArmed blocks until after() has been called with duration d, +// draining (and discarding) any other durations seen along the way. +func (f *fakeAfter) waitArmed(t *testing.T, d time.Duration) { + t.Helper() + deadline := time.After(2 * time.Second) + for { + select { + case got := <-f.armed: + if got == d { + return + } + case <-deadline: + t.Fatalf("timed out waiting for a timer to be armed at %v", d) + } + } +} + +// assertNotArmed drains any pending arm notifications and fails if d is +// among them. +func (f *fakeAfter) assertNotArmed(t *testing.T, d time.Duration) { + t.Helper() + for { + select { + case got := <-f.armed: + if got == d { + t.Fatalf("timer armed at %v, want it never armed", d) + } + default: + return + } + } +} + +func newTestScheduler(t *testing.T, settings Settings, fake *fakeAfter) (*Scheduler, *Sweeper) { + t.Helper() + sweeper, _ := newTestSweeper(t) + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + scheduler := NewScheduler(sweeper.settingsStore, sweeper, SchedulerOptions{After: fake.after}) + return scheduler, sweeper +} + +func TestScheduler_RunsCensusAtStartRegardlessOfEnabled(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + // The next after() call is census's re-arm, issued only once RunCensus + // (called synchronously beforehand in run()) has returned. + fake.waitArmed(t, sweepInterval(settings)) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("office_routine_runs census state = %v, want fresh (disabled must not block the census)", counts.OfficeRoutineRuns.State) + } + if counts.Runs.State != CensusFresh || counts.RunEvents.State != CensusFresh { + t.Fatalf("census not fresh for every table: %+v", counts) + } +} + +func TestScheduler_SweepArmedAtFirstDelayWhenEnabledAtStart(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, firstSweepDelay) +} + +func TestScheduler_SweepNotArmedWhenDisabledAtStart(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, sweepInterval(settings)) // census's arm proves the loop is running + fake.assertNotArmed(t, firstSweepDelay) +} + +func TestScheduler_SweepFiresAndReArmsAtFullIntervalNotFirstDelayAgain(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + settings.SweepIntervalHours = 1 + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, firstSweepDelay) + fake.fire(t, firstSweepDelay) + + fake.waitArmed(t, sweepInterval(settings)) // re-armed at the full interval, not another 5-minute delay + + if _, ok := sweeper.LastSweepSnapshot(); !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true (the fired timer must have run a sweep)") + } +} + +func TestScheduler_EnablingFromDisabledArmsAtFirstDelay(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + enabled := settings + enabled.Enabled = true + scheduler.ApplySettings(enabled) + + fake.waitArmed(t, firstSweepDelay) +} + +func TestScheduler_SettingsChangeWhileEnabledReArmsAtNewIntervalNotFirstDelay(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + settings.SweepIntervalHours = 1 + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, firstSweepDelay) // initial arm; timer never fired + + changed := settings + changed.SweepIntervalHours = 2 + scheduler.ApplySettings(changed) + + fake.waitArmed(t, sweepInterval(changed)) // re-armed at the new interval, not firstSweepDelay again +} + +func TestScheduler_DisablingStopsArmingSweepButNotCensus(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, firstSweepDelay) + + disabled := settings + disabled.Enabled = false + scheduler.ApplySettings(disabled) + + fake.waitArmed(t, sweepInterval(disabled)) // census's re-arm proves the wake was processed + fake.assertNotArmed(t, firstSweepDelay) +} + +func TestScheduler_StartTwiceIsNoop(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("first Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("second Start: %v", err) + } + // A second run() goroutine would double-arm; draining once more must + // time out rather than find another immediate arm. + select { + case d := <-fake.armed: + t.Fatalf("second Start armed another timer at %v", d) + case <-time.After(100 * time.Millisecond): + } +} + +func TestScheduler_StopJoinsWithoutHanging(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + fake.waitArmed(t, firstSweepDelay) + + done := make(chan struct{}) + go func() { + scheduler.Stop() + close(done) + }() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("Stop did not return") + } +} diff --git a/apps/backend/internal/office/retention/settings_store.go b/apps/backend/internal/office/retention/settings_store.go new file mode 100644 index 00000000000..d7cf6d53cb5 --- /dev/null +++ b/apps/backend/internal/office/retention/settings_store.go @@ -0,0 +1,91 @@ +package retention + +import ( + "context" + "encoding/json" + "fmt" + + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// settingsKey is the internal/system/settings.Store key holding the +// retention policy document. The live key/value table is "settings", +// reached through settings.Store; system_settings is a legacy SQLite-only +// table read once for migration and is absent on PostgreSQL. +const settingsKey = "office_run_retention" + +// SettingsStore persists and reads the retention policy document. +type SettingsStore struct { + store *systemsettings.Store +} + +// NewSettingsStore wraps the shared key/value settings store. +func NewSettingsStore(store *systemsettings.Store) *SettingsStore { + return &SettingsStore{store: store} +} + +// GetSettings reads the retention policy for reporting and startup use. An +// unreadable or unparseable document falls back to DefaultSettings and +// wraps ErrInvalidPersistedSettings so the caller can raise a health issue; +// it never fails outright (AC-OFFICE-RUN-HISTORY-RETENTION-004.4). +func (s *SettingsStore) GetSettings(ctx context.Context) (Settings, error) { + raw, found, err := s.store.Get(ctx, settingsKey) + if err != nil { + return DefaultSettings(), fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + if !found { + return DefaultSettings(), nil + } + return decodeAndNormalize(raw, DefaultSettings()) +} + +// GetSettingsForSweep reads the retention policy on the writer pool at +// sweep start, per AC-OFFICE-RUN-HISTORY-RETENTION-004.5: a sweep must read +// the stored settings at its start rather than a value cached from a +// notification, so a backend that did not serve the write still sweeps +// under the new policy. Unlike GetSettings, a read or parse failure here +// returns ErrInvalidPersistedSettings with no usable Settings value — the +// caller must skip the sweep rather than fall back to the (possibly +// shorter) default window and delete history the operator configured the +// system to keep. A document that was never saved is the legitimate empty +// state, not a failure, and yields the defaults. +func (s *SettingsStore) GetSettingsForSweep(ctx context.Context) (Settings, error) { + raw, found, err := s.store.GetConsistent(ctx, settingsKey) + if err != nil { + return Settings{}, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + if !found { + return DefaultSettings(), nil + } + return decodeAndNormalize(raw, Settings{}) +} + +func decodeAndNormalize(raw []byte, fallback Settings) (Settings, error) { + var doc Settings + if err := json.Unmarshal(raw, &doc); err != nil { + return fallback, fmt.Errorf("%w: decode JSON: %w", ErrInvalidPersistedSettings, err) + } + normalized, err := NormalizeSettings(doc) + if err != nil { + return fallback, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + return normalized, nil +} + +// SaveSettings normalizes and persists a full settings document, replacing +// whatever was stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.9). A rejected +// write leaves the stored document unchanged. +func (s *SettingsStore) SaveSettings(ctx context.Context, in Settings) (Settings, error) { + normalized, err := NormalizeSettings(in) + if err != nil { + return Settings{}, err + } + raw, err := json.Marshal(normalized) + if err != nil { + return Settings{}, fmt.Errorf("encode retention settings: %w", err) + } + if err := s.store.Save(ctx, settingsKey, raw); err != nil { + return Settings{}, err + } + return normalized, nil +} diff --git a/apps/backend/internal/office/retention/settings_store_test.go b/apps/backend/internal/office/retention/settings_store_test.go new file mode 100644 index 00000000000..eacfb89e789 --- /dev/null +++ b/apps/backend/internal/office/retention/settings_store_test.go @@ -0,0 +1,154 @@ +package retention + +import ( + "context" + "errors" + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +func newTestSettingsStore(t *testing.T) (*SettingsStore, *systemsettings.Store) { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + raw, err := systemsettings.NewStore(db.NewPool(conn, conn)) + if err != nil { + t.Fatalf("new settings store: %v", err) + } + return NewSettingsStore(raw), raw +} + +func TestSettingsStore_MissingReturnsDefaults(t *testing.T) { + store, _ := newTestSettingsStore(t) + got, err := store.GetSettings(context.Background()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettings() = %+v, want defaults", got) + } +} + +func TestSettingsStore_SaveThenGetRoundTrips(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.SweepIntervalHours = 12 + in.RoutineRuns.WindowDays = 90 + + saved, err := store.SaveSettings(ctx, in) + if err != nil { + t.Fatalf("SaveSettings: %v", err) + } + if saved != in { + t.Fatalf("SaveSettings returned %+v, want %+v", saved, in) + } + + got, err := store.GetSettings(ctx) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got != in { + t.Fatalf("GetSettings() = %+v, want %+v", got, in) + } +} + +func TestSettingsStore_SaveRejectsOutOfRangeAndLeavesStoredUnchanged(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.SweepIntervalHours = 12 + if _, err := store.SaveSettings(ctx, in); err != nil { + t.Fatalf("seed SaveSettings: %v", err) + } + + bad := DefaultSettings() + bad.BatchLimit = 1 + if _, err := store.SaveSettings(ctx, bad); !errors.Is(err, ErrValidation) { + t.Fatalf("SaveSettings(bad): err = %v, want ErrValidation", err) + } + + got, err := store.GetSettings(ctx) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got.SweepIntervalHours != 12 { + t.Fatalf("stored settings changed after rejected write: %+v", got) + } +} + +// TestSettingsStore_UnparseableFallsBackToDefaultsForReporting proves +// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: reading for reporting/startup +// tolerates an unreadable document and yields the documented defaults +// (wrapped in ErrInvalidPersistedSettings so a caller can raise a health +// issue), never failing outright. +func TestSettingsStore_UnparseableFallsBackToDefaultsForReporting(t *testing.T) { + store, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + got, err := store.GetSettings(ctx) + if !errors.Is(err, ErrInvalidPersistedSettings) { + t.Fatalf("GetSettings: err = %v, want ErrInvalidPersistedSettings", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettings() = %+v, want defaults on unparseable document", got) + } +} + +// TestSettingsStore_ForSweepFailsClosedOnUnparseable proves the other half +// of AC-OFFICE-RUN-HISTORY-RETENTION-004.5: reading settings *to delete by* +// must not silently fall back to the (possibly shorter) default window. The +// caller sees ErrInvalidPersistedSettings and a zero Settings value, and is +// expected to skip the sweep rather than run it under defaults. +func TestSettingsStore_ForSweepFailsClosedOnUnparseable(t *testing.T) { + store, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + _, err := store.GetSettingsForSweep(ctx) + if !errors.Is(err, ErrInvalidPersistedSettings) { + t.Fatalf("GetSettingsForSweep: err = %v, want ErrInvalidPersistedSettings", err) + } +} + +func TestSettingsStore_ForSweepMissingUsesDefaults(t *testing.T) { + store, _ := newTestSettingsStore(t) + got, err := store.GetSettingsForSweep(context.Background()) + if err != nil { + t.Fatalf("GetSettingsForSweep: %v", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettingsForSweep() = %+v, want defaults when nothing stored", got) + } +} + +func TestSettingsStore_ForSweepReadsWriterPool(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.RoutineRuns.WindowDays = 3650 + if _, err := store.SaveSettings(ctx, in); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + got, err := store.GetSettingsForSweep(ctx) + if err != nil { + t.Fatalf("GetSettingsForSweep: %v", err) + } + if got.RoutineRuns.WindowDays != 3650 { + t.Fatalf("GetSettingsForSweep() = %+v, want the just-saved window", got) + } +} diff --git a/apps/backend/internal/office/retention/settings_wire.go b/apps/backend/internal/office/retention/settings_wire.go new file mode 100644 index 00000000000..e82f76af7ba --- /dev/null +++ b/apps/backend/internal/office/retention/settings_wire.go @@ -0,0 +1,169 @@ +package retention + +import ( + "bytes" + "encoding/json" + "fmt" +) + +// retentionSettingsEnvelope captures each top-level field as raw JSON +// rather than a typed value, so a request body can be told apart into +// three cases per field: absent (nil RawMessage, take the documented +// default), present as the literal JSON null (rejected — +// AC-OFFICE-RUN-HISTORY-RETENTION-004.9), or present with a value (decoded +// and validated). A plain typed struct cannot distinguish the first two: an +// ordinary `*int` field is nil either way. +type retentionSettingsEnvelope struct { + Enabled json.RawMessage `json:"enabled"` + SweepIntervalHours json.RawMessage `json:"sweep_interval_hours"` + BatchLimit json.RawMessage `json:"batch_limit"` + RoutineRuns json.RawMessage `json:"routine_runs"` + Runs json.RawMessage `json:"runs"` + RunEvents json.RawMessage `json:"run_events"` +} + +type tableSettingsEnvelope struct { + WindowDays json.RawMessage `json:"window_days"` + FloorPerOwner json.RawMessage `json:"floor_per_owner"` + WarnRows json.RawMessage `json:"warn_rows"` +} + +type runEventsSettingsEnvelope struct { + WarnRows json.RawMessage `json:"warn_rows"` +} + +// decodeRetentionSettings implements AC-OFFICE-RUN-HISTORY-RETENTION-004.9's +// PUT semantics: a full replace where an omitted field takes its documented +// default, an unrecognized field or an explicit null is rejected naming the +// field, and nothing is written on any rejection (the caller is expected to +// not persist the zero-value Settings returned alongside a non-nil error). +// Range validation (AC-004.3) is NormalizeSettings's job, called by +// SettingsStore.SaveSettings after this decode succeeds. +func decodeRetentionSettings(body []byte) (Settings, error) { + defaults := DefaultSettings() + + dec := json.NewDecoder(bytes.NewReader(body)) + dec.DisallowUnknownFields() + var env retentionSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return Settings{}, err + } + + enabled, err := decodeBoolField(env.Enabled, "enabled", defaults.Enabled) + if err != nil { + return Settings{}, err + } + sweepIntervalHours, err := decodeIntField(env.SweepIntervalHours, "sweep_interval_hours", defaults.SweepIntervalHours) + if err != nil { + return Settings{}, err + } + batchLimit, err := decodeIntField(env.BatchLimit, "batch_limit", defaults.BatchLimit) + if err != nil { + return Settings{}, err + } + routineRuns, err := decodeTableSettings(env.RoutineRuns, "routine_runs", defaults.RoutineRuns) + if err != nil { + return Settings{}, err + } + runs, err := decodeTableSettings(env.Runs, "runs", defaults.Runs) + if err != nil { + return Settings{}, err + } + runEvents, err := decodeRunEventsSettings(env.RunEvents, "run_events", defaults.RunEvents) + if err != nil { + return Settings{}, err + } + + return Settings{ + Enabled: enabled, + SweepIntervalHours: sweepIntervalHours, + BatchLimit: batchLimit, + RoutineRuns: routineRuns, + Runs: runs, + RunEvents: runEvents, + }, nil +} + +func decodeTableSettings(raw json.RawMessage, name string, defaults TableSettings) (TableSettings, error) { + if raw == nil { + return defaults, nil + } + if isJSONNull(raw) { + return TableSettings{}, rejectNull(name) + } + dec := json.NewDecoder(bytes.NewReader(raw)) + dec.DisallowUnknownFields() + var env tableSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return TableSettings{}, fmt.Errorf("%s: %w", name, err) + } + windowDays, err := decodeIntField(env.WindowDays, name+".window_days", defaults.WindowDays) + if err != nil { + return TableSettings{}, err + } + floorPerOwner, err := decodeIntField(env.FloorPerOwner, name+".floor_per_owner", defaults.FloorPerOwner) + if err != nil { + return TableSettings{}, err + } + warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows) + if err != nil { + return TableSettings{}, err + } + return TableSettings{WindowDays: windowDays, FloorPerOwner: floorPerOwner, WarnRows: warnRows}, nil +} + +func decodeRunEventsSettings(raw json.RawMessage, name string, defaults RunEventsSettings) (RunEventsSettings, error) { + if raw == nil { + return defaults, nil + } + if isJSONNull(raw) { + return RunEventsSettings{}, rejectNull(name) + } + dec := json.NewDecoder(bytes.NewReader(raw)) + dec.DisallowUnknownFields() + var env runEventsSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return RunEventsSettings{}, fmt.Errorf("%s: %w", name, err) + } + warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows) + if err != nil { + return RunEventsSettings{}, err + } + return RunEventsSettings{WarnRows: warnRows}, nil +} + +func decodeBoolField(raw json.RawMessage, name string, defaultVal bool) (bool, error) { + if raw == nil { + return defaultVal, nil + } + if isJSONNull(raw) { + return false, rejectNull(name) + } + var v bool + if err := json.Unmarshal(raw, &v); err != nil { + return false, fmt.Errorf("%s: %w", name, err) + } + return v, nil +} + +func decodeIntField(raw json.RawMessage, name string, defaultVal int) (int, error) { + if raw == nil { + return defaultVal, nil + } + if isJSONNull(raw) { + return 0, rejectNull(name) + } + var v int + if err := json.Unmarshal(raw, &v); err != nil { + return 0, fmt.Errorf("%s: %w", name, err) + } + return v, nil +} + +func isJSONNull(raw json.RawMessage) bool { + return bytes.Equal(bytes.TrimSpace(raw), []byte("null")) +} + +func rejectNull(name string) error { + return fmt.Errorf("%s: must not be null; omit the field to use its default", name) +} diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go new file mode 100644 index 00000000000..eaa1a879d0d --- /dev/null +++ b/apps/backend/internal/office/retention/store.go @@ -0,0 +1,380 @@ +package retention + +import ( + "context" + "database/sql" + "fmt" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/db/dialect" +) + +// queryer is the subset of *sqlx.DB / *sqlx.Conn this package needs to run +// a sweep or a census. The sweep runs every statement for its whole +// duration through one queryer: the writer pool directly on SQLite, or one +// dedicated connection on PostgreSQL (see lock.go) — so the advisory lock +// and the batches it protects can never diverge onto different sessions. +type queryer interface { + db.Rebinder + ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error) + QueryxContext(ctx context.Context, query string, args ...any) (*sqlx.Rows, error) + QueryRowxContext(ctx context.Context, query string, args ...any) *sqlx.Row + GetContext(ctx context.Context, dest any, query string, args ...any) error + SelectContext(ctx context.Context, dest any, query string, args ...any) error + BeginTxx(ctx context.Context, opts *sql.TxOptions) (*sqlx.Tx, error) +} + +// Store is the raw SQL access layer for retention: the eligibility counts +// (used by both the preview and the backlog check), the batch deletes, and +// the status census. It holds no state of its own. +type Store struct { + pool *db.Pool +} + +// NewStore wraps the database pool office_routine_runs, runs, and their +// satellites live in. +func NewStore(pool *db.Pool) *Store { + return &Store{pool: pool} +} + +// IsPostgres reports whether the writer pool is PostgreSQL. On SQLite one +// backend process owns the database file, so the in-process sweeping guard +// is sufficient and no advisory lock is taken. +func (s *Store) IsPostgres() bool { + return dialect.IsPostgres(s.pool.Writer().DriverName()) +} + +func routineRunEligibleSubquery() string { + return ` + SELECT id, + COALESCE(completed_at, created_at) AS completion_time, + ROW_NUMBER() OVER ( + PARTITION BY routine_id + ORDER BY COALESCE(completed_at, created_at) DESC, id DESC + ) AS rn + FROM office_routine_runs + WHERE status IN (?)` +} + +func runEligibleSubquery() string { + return ` + SELECT id, + COALESCE(finished_at, requested_at) AS completion_time, + ROW_NUMBER() OVER ( + PARTITION BY agent_profile_id + ORDER BY COALESCE(finished_at, requested_at) DESC, id DESC + ) AS rn + FROM runs + WHERE status IN (?)` +} + +// CountEligibleRoutineRuns is the office_routine_runs eligibility count, +// uncapped by any batch limit: used for both the preview's WouldDelete +// (AC-OFFICE-RUN-HISTORY-RETENTION-003.9) and the backlog determination +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3) so the two can never disagree +// about what "eligible" means. +func (s *Store) CountEligibleRoutineRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) { + query := `SELECT COUNT(*) FROM (` + routineRunEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?` + return countEligible(ctx, q, query, RoutineRunHistoryStatuses, floor, cutoff) +} + +// CountEligibleRuns is the runs table's equivalent of CountEligibleRoutineRuns. +func (s *Store) CountEligibleRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) { + query := `SELECT COUNT(*) FROM (` + runEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?` + return countEligible(ctx, q, query, RunHistoryStatuses, floor, cutoff) +} + +func countEligible(ctx context.Context, q queryer, query string, statuses []string, floor int, cutoff time.Time) (int64, error) { + bound, args, err := db.Bind(q, query, statuses, floor, cutoff) + if err != nil { + return 0, err + } + var count int64 + if err := q.QueryRowxContext(ctx, bound, args...).Scan(&count); err != nil { + return 0, err + } + return count, nil +} + +// DeleteRoutineRunsBatch deletes at most batchLimit eligible +// office_routine_runs rows, oldest first, re-asserting status, age and the +// per-owner floor in the same statement +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3, -002.4). One statement is +// already atomic, satisfying -002.5 without an explicit transaction. +func (s *Store) DeleteRoutineRunsBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (int64, error) { + query := ` + DELETE FROM office_routine_runs + WHERE id IN ( + SELECT id FROM (` + routineRunEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ORDER BY completion_time ASC, id ASC + LIMIT ? + ) + AND status IN (?) + AND COALESCE(completed_at, created_at) < ?` + bound, args, err := db.Bind(q, query, + RoutineRunHistoryStatuses, floor, cutoff, batchLimit, + RoutineRunHistoryStatuses, cutoff, + ) + if err != nil { + return 0, err + } + res, err := q.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + return res.RowsAffected() +} + +// RunBatchResult reports what one runs batch (and its satellites) actually +// deleted. Abandoned is true only when the batch was rolled back twice in a +// row and gave up — every count is then zero, matching what the rollback +// left committed (AC-OFFICE-RUN-HISTORY-RETENTION-002.7). +type RunBatchResult struct { + RunsDeleted int64 + RunEventsDeleted int64 + RouteAttemptsDeleted int64 + RunSkillsDeleted int64 + Abandoned bool +} + +// DeleteRunBatch selects up to batchLimit eligible runs, deletes each +// selected run's satellites and then the run itself in one transaction, +// re-asserting the whole eligibility predicate (status, age, and floor) at +// delete time. If a concurrent ScheduleRetry resurrects a row between +// selection and delete, step 5's affected-row count falls short of the +// selected id count; the whole transaction is rolled back and retried once +// with a fresh selection. A second mismatch abandons the batch +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4, -002.7). +func (s *Store) DeleteRunBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (RunBatchResult, error) { + for attempt := 0; attempt < 2; attempt++ { + if testBeforeSelectEligibleRunIDs != nil { + testBeforeSelectEligibleRunIDs(attempt) + } + ids, err := s.selectEligibleRunIDs(ctx, q, cutoff, floor, batchLimit) + if err != nil { + return RunBatchResult{}, err + } + if len(ids) == 0 { + return RunBatchResult{}, nil + } + if testAfterSelectEligibleRunIDs != nil { + testAfterSelectEligibleRunIDs(attempt, ids) + } + result, matched, err := s.deleteRunBatchOnce(ctx, q, ids, cutoff, floor) + if err != nil { + return RunBatchResult{}, err + } + if matched { + return result, nil + } + } + return RunBatchResult{Abandoned: true}, nil +} + +// testBeforeSelectEligibleRunIDs and testAfterSelectEligibleRunIDs, when +// set, bracket each of DeleteRunBatch's (at most two) selections — a +// deterministic seam this package's own tests use to exercise the +// selection-to-delete resurrection race, including the two-consecutive- +// mismatches abandon path, without depending on cross-connection goroutine +// timing. Never set outside tests. +var ( + testBeforeSelectEligibleRunIDs func(attempt int) + testAfterSelectEligibleRunIDs func(attempt int, ids []string) +) + +func (s *Store) selectEligibleRunIDs(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) ([]string, error) { + query := ` + SELECT id FROM (` + runEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ORDER BY completion_time ASC, id ASC + LIMIT ?` + bound, args, err := db.Bind(q, query, RunHistoryStatuses, floor, cutoff, batchLimit) + if err != nil { + return nil, err + } + rows, err := q.QueryxContext(ctx, bound, args...) + if err != nil { + return nil, err + } + defer func() { _ = rows.Close() }() + var ids []string + for rows.Next() { + var id string + if err := rows.Scan(&id); err != nil { + return nil, err + } + ids = append(ids, id) + } + return ids, rows.Err() +} + +func (s *Store) deleteRunBatchOnce(ctx context.Context, q queryer, ids []string, cutoff time.Time, floor int) (RunBatchResult, bool, error) { + tx, err := q.BeginTxx(ctx, nil) + if err != nil { + return RunBatchResult{}, false, err + } + committed := false + defer func() { + if !committed { + _ = tx.Rollback() + } + }() + + runEventsDeleted, err := deleteByRunIDs(ctx, tx, "run_events", ids) + if err != nil { + return RunBatchResult{}, false, err + } + routeAttemptsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_route_attempts", ids) + if err != nil { + return RunBatchResult{}, false, err + } + runSkillsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_skills", ids) + if err != nil { + return RunBatchResult{}, false, err + } + + query := ` + DELETE FROM runs + WHERE id IN (?) + AND id IN ( + SELECT id FROM (` + runEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + )` + bound, args, err := db.Bind(tx, query, ids, RunHistoryStatuses, floor, cutoff) + if err != nil { + return RunBatchResult{}, false, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return RunBatchResult{}, false, err + } + runsDeleted, err := res.RowsAffected() + if err != nil { + return RunBatchResult{}, false, err + } + if runsDeleted != int64(len(ids)) { + return RunBatchResult{}, false, nil + } + + if err := tx.Commit(); err != nil { + return RunBatchResult{}, false, err + } + committed = true + return RunBatchResult{ + RunsDeleted: runsDeleted, + RunEventsDeleted: runEventsDeleted, + RouteAttemptsDeleted: routeAttemptsDeleted, + RunSkillsDeleted: runSkillsDeleted, + }, true, nil +} + +func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) { + query := fmt.Sprintf(`DELETE FROM %s WHERE run_id IN (?)`, table) + bound, args, err := db.Bind(tx, query, ids) + if err != nil { + return 0, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + return res.RowsAffected() +} + +// CountRunEvents is run_events' plain retained count: it has no status +// column, so it keeps a bare COUNT(*) rather than a census. +func (s *Store) CountRunEvents(ctx context.Context, q queryer) (int64, error) { + var count int64 + if err := q.GetContext(ctx, &count, `SELECT COUNT(*) FROM run_events`); err != nil { + return 0, err + } + return count, nil +} + +// CensusRoutineRuns issues office_routine_runs' status census +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10, -003.5, -003.11): one +// GROUP BY status scan yields both the retained count (the sum across +// every status) and the unknown-status detector, since the sweep's own +// predicate selects only history statuses and therefore cannot observe a +// status outside it. AC-003.5's top-routine attribution needs a +// routine_id dimension the status census does not have, so it is a +// second aggregation rather than reusing the first result set. +func (s *Store) CensusRoutineRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + statusCounts, err := statusCensus(ctx, q, "office_routine_runs") + if err != nil { + return TableCensus{}, err + } + retained, unknown := summarizeStatusCensus(statusCounts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + census := TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now} + if retained > 0 { + topID, topCount, err := topRoutineByRetainedRows(ctx, q) + if err != nil { + return TableCensus{}, err + } + census.TopRoutineID = topID + census.TopRoutineShare = float64(topCount) / float64(retained) + } + return census, nil +} + +// CensusRuns issues runs' status census, the runs-table equivalent of +// CensusRoutineRuns without the routine attribution AC-003.5 is specific +// to office_routine_runs. +func (s *Store) CensusRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + statusCounts, err := statusCensus(ctx, q, "runs") + if err != nil { + return TableCensus{}, err + } + retained, unknown := summarizeStatusCensus(statusCounts, RunHistoryStatuses, RunLiveStatuses) + return TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now}, nil +} + +// CensusRunEvents is run_events' census: a plain count, since the table +// has no status column and therefore no unknown-status detection. +func (s *Store) CensusRunEvents(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + count, err := s.CountRunEvents(ctx, q) + if err != nil { + return TableCensus{}, err + } + return TableCensus{RetainedCount: count, AsOf: now}, nil +} + +func statusCensus(ctx context.Context, q queryer, table string) (map[string]int64, error) { + query := fmt.Sprintf(`SELECT status, COUNT(*) AS count FROM %s GROUP BY status`, table) + rows, err := q.QueryxContext(ctx, query) + if err != nil { + return nil, err + } + defer func() { _ = rows.Close() }() + counts := map[string]int64{} + for rows.Next() { + var status string + var count int64 + if err := rows.Scan(&status, &count); err != nil { + return nil, err + } + counts[status] = count + } + return counts, rows.Err() +} + +func topRoutineByRetainedRows(ctx context.Context, q queryer) (string, int64, error) { + var row struct { + RoutineID string `db:"routine_id"` + Retained int64 `db:"retained"` + } + query := ` + SELECT routine_id, COUNT(*) AS retained + FROM office_routine_runs + GROUP BY routine_id + ORDER BY retained DESC, routine_id ASC + LIMIT 1` + if err := q.GetContext(ctx, &row, query); err != nil { + return "", 0, err + } + return row.RoutineID, row.Retained, nil +} diff --git a/apps/backend/internal/office/retention/store_census_test.go b/apps/backend/internal/office/retention/store_census_test.go new file mode 100644 index 00000000000..2b140d9dc22 --- /dev/null +++ b/apps/backend/internal/office/retention/store_census_test.go @@ -0,0 +1,159 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/kandev/kandev/internal/db" +) + +func TestCensusRoutineRuns_RetainedCountIsSumOfEveryStatus(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "skipped", timePtr(daysAgo(2)), daysAgo(2)) + seedRoutineRun(t, conn, newID(), "r-1", "received", nil, daysAgo(0)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3", census.RetainedCount) + } + if len(census.UnknownStatuses) != 0 { + t.Fatalf("unknownStatuses = %v, want none", census.UnknownStatuses) + } +} + +func TestCensusRoutineRuns_EmptyTableReturnsZeroNoTopRoutine(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 0 { + t.Fatalf("retainedCount = %d, want 0", census.RetainedCount) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID) + } +} + +func TestCensusRoutineRuns_DetectsUnknownStatus(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0] != "quarantined" { + t.Fatalf("unknownStatuses = %v, want [quarantined]", census.UnknownStatuses) + } +} + +func TestCensusRoutineRuns_AttributesTopRoutineByRetainedShare(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-heavy") + seedRoutine(t, conn, "r-light") + for i := 0; i < 3; i++ { + seedRoutineRun(t, conn, newID(), "r-heavy", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + seedRoutineRun(t, conn, newID(), "r-light", "done", timePtr(daysAgo(1)), daysAgo(1)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 4 { + t.Fatalf("retainedCount = %d, want 4", census.RetainedCount) + } + if census.TopRoutineID != "r-heavy" { + t.Fatalf("topRoutineID = %q, want r-heavy", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 0.75; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } +} + +func TestCensusRoutineRuns_TiesAttributeToLowerRoutineID(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-b") + seedRoutine(t, conn, "r-a") + seedRoutineRun(t, conn, newID(), "r-b", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-a", "done", timePtr(daysAgo(1)), daysAgo(1)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.TopRoutineID != "r-a" { + t.Fatalf("topRoutineID = %q, want r-a (lower id on tie)", census.TopRoutineID) + } +} + +func TestCensusRuns_RetainedCountIsSumOfEveryStatusNoTopAttribution(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRun(t, conn, newID(), "agent-1", "queued", nil, daysAgo(0)) + seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0)) + + census, err := store.CensusRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRuns: %v", err) + } + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3", census.RetainedCount) + } + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0] != "mystery" { + t.Fatalf("unknownStatuses = %v, want [mystery]", census.UnknownStatuses) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty (runs has no routine attribution)", census.TopRoutineID) + } +} + +func TestCensusRunEvents_PlainCountNoStatusDetection(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRunEvent(t, conn, runID, 1) + seedRunEvent(t, conn, runID, 2) + + census, err := store.CensusRunEvents(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRunEvents: %v", err) + } + if census.RetainedCount != 2 { + t.Fatalf("retainedCount = %d, want 2", census.RetainedCount) + } + if census.UnknownStatuses != nil { + t.Fatalf("unknownStatuses = %v, want nil", census.UnknownStatuses) + } +} + +func timePtr(t time.Time) *time.Time { return &t } diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go new file mode 100644 index 00000000000..3c34d44afb3 --- /dev/null +++ b/apps/backend/internal/office/retention/store_test.go @@ -0,0 +1,428 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/google/uuid" + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// testDB builds a fresh in-memory SQLite database carrying the real office +// schema (including the retention indexes), so eligibility queries run +// against the genuine table shapes rather than a hand-rolled fixture. +func testDB(t *testing.T) *sqlx.DB { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + return conn +} + +func seedRoutine(t *testing.T, conn *sqlx.DB, id string) { + t.Helper() + now := time.Now().UTC() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routines (id, workspace_id, name, created_at, updated_at) + VALUES (?, 'ws-1', ?, ?, ?) + `), id, id, now, now); err != nil { + t.Fatalf("seed routine %s: %v", id, err) + } +} + +func seedRoutineRun(t *testing.T, conn *sqlx.DB, id, routineID, status string, completedAt *time.Time, createdAt time.Time) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at) + VALUES (?, ?, 'trigger', ?, ?, ?) + `), id, routineID, status, completedAt, createdAt); err != nil { + t.Fatalf("seed routine run %s: %v", id, err) + } +} + +func seedRun(t *testing.T, conn *sqlx.DB, id, agentProfileID, status string, finishedAt *time.Time, requestedAt time.Time) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO runs (id, agent_profile_id, reason, status, requested_at, finished_at) + VALUES (?, ?, 'test', ?, ?, ?) + `), id, agentProfileID, status, requestedAt, finishedAt); err != nil { + t.Fatalf("seed run %s: %v", id, err) + } +} + +func seedRunEvent(t *testing.T, conn *sqlx.DB, runID string, seq int) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO run_events (run_id, seq, event_type, created_at) + VALUES (?, ?, 'test', ?) + `), runID, seq, time.Now().UTC()); err != nil { + t.Fatalf("seed run event for %s: %v", runID, err) + } +} + +func seedRouteAttempt(t *testing.T, conn *sqlx.DB, runID string, seq int) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_run_route_attempts (run_id, seq, provider_id, model, tier, outcome, started_at) + VALUES (?, ?, 'p', 'm', 't', 'ok', ?) + `), runID, seq, time.Now().UTC()); err != nil { + t.Fatalf("seed route attempt for %s: %v", runID, err) + } +} + +func countRows(t *testing.T, conn *sqlx.DB, query string, args ...any) int64 { + t.Helper() + var n int64 + if err := conn.Get(&n, conn.Rebind(query), args...); err != nil { + t.Fatalf("count query %q: %v", query, err) + } + return n +} + +func daysAgo(n int) time.Time { return time.Now().UTC().AddDate(0, 0, -n) } + +func newID() string { return uuid.New().String() } + +func TestCountEligibleRoutineRuns_RespectsStatusWindowAndFloor(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + // 3 history rows older than the window, floor 50: all inside the floor, + // none eligible. + for i := 0; i < 3; i++ { + completed := daysAgo(40 + i) + seedRoutineRun(t, conn, newID(), routineID, "coalesced", &completed, completed) + } + // A task_created (live-state) row, ancient: never eligible regardless of + // age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). + ancient := daysAgo(3650) + seedRoutineRun(t, conn, newID(), routineID, "task_created", nil, ancient) + + count, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if count != 0 { + t.Fatalf("count = %d, want 0 (all 3 history rows within the floor of 50)", count) + } + + // Add 50 more, all older than window: total history rows now 53, floor + // 50, so exactly 3 are eligible (the 3 oldest, since floor keeps the + // newest 50 of the 53). + for i := 0; i < 50; i++ { + completed := daysAgo(35 + i) + seedRoutineRun(t, conn, newID(), routineID, "done", &completed, completed) + } + count, err = store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if count != 3 { + t.Fatalf("count = %d, want 3 (53 history rows, floor 50)", count) + } +} + +func TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + var oldest, middle, newest string + oldest, middle, newest = newID(), newID(), newID() + oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45) + seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC) + seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC) + seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC) + + // floor 0 so all three are eligible; batch limit 2 -> the two oldest go, + // the newest survives as backlog. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 2 { + t.Fatalf("deleted = %d, want 2", deleted) + } + remaining := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest) + if remaining != 1 { + t.Fatalf("newest row was deleted; oldest-first ordering violated") + } + for _, id := range []string{oldest, middle} { + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 { + t.Fatalf("row %s (older) still present after batch limit 2", id) + } + } +} + +func TestDeleteRoutineRunsBatch_FloorReassertedAtDeleteTime(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + c := daysAgo(90) + id := newID() + seedRoutineRun(t, conn, id, routineID, "done", &c, c) + + // Floor 1 (>= the single row present) means the row is protected: it + // is the newest (and only) row for its routine. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 1, 100) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 0 { + t.Fatalf("deleted = %d, want 0 (row is inside the floor)", deleted) + } +} + +func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + seedRunEvent(t, conn, runID, 1) + seedRouteAttempt(t, conn, runID, 0) + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 1 || result.RunEventsDeleted != 2 || result.RouteAttemptsDeleted != 1 { + t.Fatalf("result = %+v, want RunsDeleted=1 RunEventsDeleted=2 RouteAttemptsDeleted=1", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d run_events rows remain referencing a deleted run", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_route_attempts WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d route attempt rows remain referencing a deleted run", n) + } +} + +// TestDeleteRunBatch_SurvivingRunKeepsEveryEvent proves +// AC-OFFICE-RUN-HISTORY-RETENTION-001.6: run_events is never deleted for a +// run that is not itself being deleted in the same transaction, no matter +// how old those events are. +func TestDeleteRunBatch_SurvivingRunKeepsEveryEvent(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + liveRunID := newID() + // queued: live state, never eligible at any age. + seedRun(t, conn, liveRunID, "agent-1", "queued", nil, daysAgo(400)) + for i := 0; i < 5; i++ { + seedRunEvent(t, conn, liveRunID, i) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 { + t.Fatalf("RunsDeleted = %d, want 0 (queued run is live state)", result.RunsDeleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, liveRunID); n != 5 { + t.Fatalf("run_events for a surviving run = %d, want 5 untouched", n) + } +} + +// TestDeleteRunBatch_ResurrectedRunSurvivesAndKeepsSatellites drives the +// AC-OFFICE-RUN-HISTORY-RETENTION-002.4 race directly: a run selected as +// eligible is resurrected to "queued" (as ScheduleRetry does, clearing +// finished_at) after selection but before the delete statement runs. The +// re-assertion inside DeleteRunBatch's delete step must catch this, +// rolling back and leaving the run and its satellites intact. +func TestDeleteRunBatch_ResurrectedRunSurvivesAndKeepsSatellites(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + // Simulate the resurrection race by racing a resurrecting UPDATE + // against DeleteRunBatch using a second connection to the same + // in-memory database (SQLite's single-writer serializes them, but the + // delete's re-assertion inside the transaction is what must catch the + // now-live row regardless of interleaving — resurrecting up front is + // the deterministic way to exercise that same code path without + // depending on goroutine scheduling). + if _, err := conn.Exec(conn.Rebind(` + UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ? + `), runID); err != nil { + t.Fatalf("resurrect run: %v", err) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op (resurrected run was never selected)", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted") + } +} + +// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives drives +// the retry path deterministically via testAfterSelectEligibleRunIDs: the +// row is eligible at selection time (so it enters attempt 0's id list), +// resurrected by the hook immediately after that selection (before +// deleteRunBatchOnce's own DELETE runs), so the re-assertion inside the +// transaction detects the mismatch, rolls back, and attempt 1 re-selects +// against the now-live row and finds nothing to do. The run and its +// satellite survive throughout. +func TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + t.Cleanup(func() { testAfterSelectEligibleRunIDs = nil }) + testAfterSelectEligibleRunIDs = func(attempt int, ids []string) { + if attempt != 0 { + return + } + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`, + ), runID); err != nil { + t.Fatalf("resurrect run mid-transaction: %v", err) + } + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op after the retry re-selects nothing", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted despite the retry") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted despite the retry") + } +} + +// TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches forces the +// resurrection race to land on *both* attempts via +// testAfterSelectEligibleRunIDs: the row is repeatedly resurrected right +// after each selection, so both attempts' delete re-assertions mismatch. +// DeleteRunBatch must give up rather than loop forever, reporting the +// batch as Abandoned — which the sweep records as that table's failure, +// not backlog (AC-OFFICE-RUN-HISTORY-RETENTION-002.7). +func TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + beforeCalls, afterCalls := 0, 0 + // Before each attempt's selection, make sure the row reads terminal + // again (undoing the previous attempt's resurrection) so it is + // selected every time, not just on attempt 0. + testBeforeSelectEligibleRunIDs = func(attempt int) { + beforeCalls++ + if attempt == 0 { + return // already terminal from seeding + } + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`, + ), finished, runID); err != nil { + t.Fatalf("re-terminalize run before attempt %d: %v", attempt, err) + } + } + // After each attempt's selection (which just proved the row was + // terminal), flip it live so that attempt's own delete re-assertion + // mismatches. + testAfterSelectEligibleRunIDs = func(attempt int, ids []string) { + afterCalls++ + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`, + ), runID); err != nil { + t.Fatalf("resurrect run mid-transaction (attempt %d): %v", attempt, err) + } + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if !result.Abandoned { + t.Fatalf("result = %+v, want Abandoned=true after two consecutive mismatches", result) + } + if result.RunsDeleted != 0 || result.RunEventsDeleted != 0 { + t.Fatalf("result = %+v, want every count zero on an abandoned batch", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("run was deleted despite an abandoned batch") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("run_events was deleted despite an abandoned batch") + } + if beforeCalls != 2 || afterCalls != 2 { + t.Fatalf("before/after hooks invoked %d/%d times, want exactly 2/2 (one per attempt)", beforeCalls, afterCalls) + } +} + +func TestCountRunEvents_PlainCount(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + runID := newID() + finished := daysAgo(1) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + for i := 0; i < 4; i++ { + seedRunEvent(t, conn, runID, i) + } + count, err := store.CountRunEvents(ctx, conn) + if err != nil { + t.Fatalf("CountRunEvents: %v", err) + } + if count != 4 { + t.Fatalf("CountRunEvents() = %d, want 4", count) + } +} diff --git a/apps/backend/internal/office/retention/sweep.go b/apps/backend/internal/office/retention/sweep.go new file mode 100644 index 00000000000..76134a76ffd --- /dev/null +++ b/apps/backend/internal/office/retention/sweep.go @@ -0,0 +1,322 @@ +package retention + +import ( + "context" + "sync" + "time" + + "github.com/kandev/kandev/internal/db" +) + +// TableSweepResult is one reported table's outcome for one sweep +// (AC-OFFICE-RUN-HISTORY-RETENTION-004.6). Backlog is only ever true for a +// swept table's own SweptTableResult; a satellite table's Backlog stays +// false because its deletion is never independently batch-limited — it +// deletes exactly the run ids its owning runs batch selected. +type TableSweepResult struct { + Deleted int64 `json:"deleted"` + Backlog bool `json:"backlog"` + Err string `json:"error"` +} + +// SweptTableResult adds preview reporting to a swept table's outcome. +// Previewed is true only when THIS sweep was a preview pass for this +// table, never a running total. +type SweptTableResult struct { + TableSweepResult + Previewed bool `json:"previewed"` + WouldDelete int64 `json:"would_delete"` +} + +// LastSweep is the in-memory value replaced wholesale at the end of each +// sweep that actually ran (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7). +// Nothing here is persisted. +type LastSweep struct { + StartedAt time.Time `json:"started_at"` + FinishedAt time.Time `json:"finished_at"` + + OfficeRoutineRuns SweptTableResult `json:"office_routine_runs"` + Runs SweptTableResult `json:"runs"` + RunEvents TableSweepResult `json:"run_events"` + RouteAttempts TableSweepResult `json:"route_attempts"` + RunSkills TableSweepResult `json:"run_skills"` +} + +type satelliteResults struct { + RunEvents TableSweepResult + RouteAttempts TableSweepResult + RunSkills TableSweepResult +} + +// Sweeper runs one sweep or one census pass at a time; the scheduler +// (scheduler.go) owns when to call each. Findings resolved here, as +// documented Build decisions: +// +// - F28 (lost-exclusivity outcome): a lock lost between tables abandons +// the whole sweep attempt as a skip — the same outcome as a local +// concurrent-sweep collision — rather than inventing a per-table +// "skipped" state the design's eight-id health catalogue has nowhere +// to report. Batches already committed on PostgreSQL stay committed +// (AC-002.5); LastSweep simply is not replaced by this attempt, and +// the next scheduled sweep reports the tables' true state either way. +// - The skip counter increments for a collision, a failed settings +// re-read, a failed lock acquisition, and a lock lost mid-sweep — every +// case where a sweep was due and did not produce a result. It does +// NOT increment when retention is disabled (AC-002.8's "run no sweep" +// is a deliberate, steady-state condition, not a due sweep that could +// not run; incrementing forever while off would make the counter +// meaningless). +type Sweeper struct { + pool *db.Pool + store *Store + settingsStore *SettingsStore + previewMarker *PreviewMarkerStore + census *CensusTracker + + mu sync.Mutex + sweeping bool + lastSweep *LastSweep + skipCount int64 + lastSkip time.Time +} + +// NewSweeper wires the sweep orchestration to its dependencies. +func NewSweeper(pool *db.Pool, store *Store, settingsStore *SettingsStore, previewMarker *PreviewMarkerStore) *Sweeper { + return &Sweeper{ + pool: pool, + store: store, + settingsStore: settingsStore, + previewMarker: previewMarker, + census: NewCensusTracker(), + } +} + +// LastSweepSnapshot returns the most recent completed sweep's result. ok is +// false before the first sweep ever completes (AC-OFFICE-RUN-HISTORY-RETENTION-004.7). +func (s *Sweeper) LastSweepSnapshot() (LastSweep, bool) { + s.mu.Lock() + defer s.mu.Unlock() + if s.lastSweep == nil { + return LastSweep{}, false + } + return *s.lastSweep, true +} + +// SkipSnapshot returns the running skip count and the last skip's time. +func (s *Sweeper) SkipSnapshot() (count int64, lastAt time.Time) { + s.mu.Lock() + defer s.mu.Unlock() + return s.skipCount, s.lastSkip +} + +// CensusSnapshot returns the current per-table retained counts. +func (s *Sweeper) CensusSnapshot() RetainedCounts { + return s.census.Snapshot() +} + +// RunSweep performs at most one sweep: office_routine_runs, then runs with +// its satellites, in that fixed order (AC-OFFICE-RUN-HISTORY-RETENTION-002.6). +func (s *Sweeper) RunSweep(ctx context.Context) { + if !s.beginSweep() { + s.recordSkip() + return + } + defer s.endSweep() + + settings, err := s.settingsStore.GetSettingsForSweep(ctx) + if err != nil { + // Settings unreadable at sweep start fails closed: no sweep, no + // fallback to the (possibly shorter) default window. + s.recordSkip() + return + } + if !settings.Enabled { + // AC-002.8: disabled means no sweep at all, not a recorded skip. + return + } + + q, session, ok := s.acquireQueryer(ctx) + if !ok { + s.recordSkip() + return + } + if session != nil { + defer session.release() + } + + now := time.Now().UTC() + report := LastSweep{StartedAt: now} + report.OfficeRoutineRuns = s.sweepRoutineRuns(ctx, q, settings.RoutineRuns, now, settings.BatchLimit) + + if testBetweenTablesSweep != nil { + testBetweenTablesSweep(q) + } + if session != nil && !session.alive(ctx) { + // F25/F28: exclusivity was lost after the first table's work. Do + // not start the second table, and do not publish a partial + // result — this whole attempt is a skip, exactly as if the lock + // had never been acquired. + s.recordSkip() + return + } + + runsResult, satellites := s.sweepRuns(ctx, q, settings.Runs, now, settings.BatchLimit) + report.Runs = runsResult + report.RunEvents = satellites.RunEvents + report.RouteAttempts = satellites.RouteAttempts + report.RunSkills = satellites.RunSkills + + report.FinishedAt = time.Now().UTC() + s.mu.Lock() + s.lastSweep = &report + s.mu.Unlock() + + incSweepCompleted() + incDeleted(TableOfficeRoutineRuns, report.OfficeRoutineRuns.Deleted) + incDeleted(TableRuns, report.Runs.Deleted) + incDeleted(TableRunEvents, report.RunEvents.Deleted) + incDeleted("office_run_route_attempts", report.RouteAttempts.Deleted) + incDeleted("office_run_skills", report.RunSkills.Deleted) +} + +// testBetweenTablesSweep, when set, runs right after office_routine_runs' +// table work and right before the alive() liveness check and the runs +// table — a deterministic seam for exercising AC-OFFICE-RUN-HISTORY-RETENTION-002.12's +// "verifies the lock connection is still alive between tables" path (F25) +// without depending on real cross-process timing. It receives the sweep's +// own queryer so a test can run diagnostics (or a second sweep attempt) +// against the exact connection in use. Never set outside tests. +var testBetweenTablesSweep func(q queryer) + +// RunCensus evaluates the retained-count census for every thresholded +// table. Read-only, so it needs no advisory lock: every backend computes +// and serves its own local view (F35). +func (s *Sweeper) RunCensus(ctx context.Context) { + q := s.pool.Reader() + now := time.Now().UTC() + + routineCensus, err := s.store.CensusRoutineRuns(ctx, q, now) + s.census.RecordRoutineRuns(routineCensus, err) + incCensus(TableOfficeRoutineRuns, err) + + runsCensus, err := s.store.CensusRuns(ctx, q, now) + s.census.RecordRuns(runsCensus, err) + incCensus(TableRuns, err) + + runEventsCensus, err := s.store.CensusRunEvents(ctx, q, now) + s.census.RecordRunEvents(runEventsCensus, err) + incCensus(TableRunEvents, err) +} + +func (s *Sweeper) beginSweep() bool { + s.mu.Lock() + defer s.mu.Unlock() + if s.sweeping { + return false + } + s.sweeping = true + return true +} + +func (s *Sweeper) endSweep() { + s.mu.Lock() + s.sweeping = false + s.mu.Unlock() +} + +func (s *Sweeper) recordSkip() { + s.mu.Lock() + defer s.mu.Unlock() + s.skipCount++ + s.lastSkip = time.Now().UTC() + incSweepSkipped() +} + +func (s *Sweeper) acquireQueryer(ctx context.Context) (queryer, *sweepSession, bool) { + if !s.store.IsPostgres() { + return s.pool.Writer(), nil, true + } + session, ok, err := acquireSweepSession(ctx, s.pool) + if err != nil || !ok { + return nil, nil, false + } + return session.queryer(), session, true +} + +// isPreviewed reads the preview marker through q, the sweep's own +// connection, rather than the shared settings pool (F27 — see the +// PreviewMarkerStore.GetWith doc comment). +func (s *Sweeper) isPreviewed(ctx context.Context, q queryer, table TableName) bool { + marker, _ := s.previewMarker.GetWith(ctx, q) + _, ok := marker[table] + return ok +} + +func (s *Sweeper) sweepRoutineRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) SweptTableResult { + if !s.isPreviewed(ctx, q, TableOfficeRoutineRuns) { + return s.previewTable(ctx, q, TableOfficeRoutineRuns, func() (int64, error) { + return s.store.CountEligibleRoutineRuns(ctx, q, now, cfg.FloorPerOwner) + }, now) + } + + eligible, err := s.store.CountEligibleRoutineRuns(ctx, q, now, cfg.FloorPerOwner) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + deleted, err := s.store.DeleteRoutineRunsBatch(ctx, q, now, cfg.FloorPerOwner, batchLimit) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + return SweptTableResult{TableSweepResult: TableSweepResult{ + Deleted: deleted, + Backlog: eligible > int64(batchLimit), + }} +} + +func (s *Sweeper) sweepRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) (SweptTableResult, satelliteResults) { + if !s.isPreviewed(ctx, q, TableRuns) { + result := s.previewTable(ctx, q, TableRuns, func() (int64, error) { + return s.store.CountEligibleRuns(ctx, q, now, cfg.FloorPerOwner) + }, now) + return result, satelliteResults{} + } + + eligible, err := s.store.CountEligibleRuns(ctx, q, now, cfg.FloorPerOwner) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} + } + result, err := s.store.DeleteRunBatch(ctx, q, now, cfg.FloorPerOwner, batchLimit) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} + } + if result.Abandoned { + // AC-002.7: an abandoned batch is that table's failure, not + // backlog; the satellites report zero, matching what the + // rollback actually left committed. + return SweptTableResult{TableSweepResult: TableSweepResult{ + Err: "batch abandoned after two consecutive mismatches", + }}, satelliteResults{} + } + return SweptTableResult{TableSweepResult: TableSweepResult{ + Deleted: result.RunsDeleted, + Backlog: eligible > int64(batchLimit), + }}, satelliteResults{ + RunEvents: TableSweepResult{Deleted: result.RunEventsDeleted}, + RouteAttempts: TableSweepResult{Deleted: result.RouteAttemptsDeleted}, + RunSkills: TableSweepResult{Deleted: result.RunSkillsDeleted}, + } +} + +// previewTable runs a table's first-ever preview pass: count eligible rows +// uncapped, mark the table previewed, and delete nothing +// (AC-OFFICE-RUN-HISTORY-RETENTION-003.2, -003.9). +func (s *Sweeper) previewTable(ctx context.Context, q queryer, table TableName, countEligible func() (int64, error), now time.Time) SweptTableResult { + count, err := countEligible() + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + if err := s.previewMarker.MarkCompletedWith(ctx, q, table, now); err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + return SweptTableResult{Previewed: true, WouldDelete: count} +} diff --git a/apps/backend/internal/office/retention/sweep_postgres_test.go b/apps/backend/internal/office/retention/sweep_postgres_test.go new file mode 100644 index 00000000000..06bfc9e70bb --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_postgres_test.go @@ -0,0 +1,201 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" + taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite" + "github.com/kandev/kandev/internal/testutil" +) + +// newPostgresTestSweeper is sweep_test.go's newTestSweeper, built against a +// real, isolated-schema PostgreSQL connection instead of in-memory SQLite. +// tasks is created first, mirroring production boot order (see +// child_summaries_postgres_test.go). +func newPostgresTestSweeper(t *testing.T, dsn string) (*Sweeper, *sqlx.DB) { + t.Helper() + conn := testutil.OpenIsolatedPostgres(t, dsn) + if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init task repo: %v", err) + } + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + store := NewStore(pool) + settingsStore := NewSettingsStore(settingsRaw) + previewMarker := NewPreviewMarkerStore(settingsRaw) + return NewSweeper(pool, store, settingsStore, previewMarker), conn +} + +// TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables is +// AC-OFFICE-RUN-HISTORY-RETENTION-002.12's mandated two-backend test: the +// seed gives the winner two tables' worth of work, and the loser's attempt +// happens in the pause between them (via testBetweenTablesSweep). A +// transaction-scoped lock would have released between the winner's two +// per-table statements and let the loser in; this must not happen with the +// session-scoped lock. +func TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + winner, conn := newPostgresTestSweeper(t, dsn) + saveZeroFloorSettings(t, winner) + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + winner.RunSweep(ctx) // preview pass for both tables; no lock contention to test yet + + loser, _ := newPostgresTestSweeper(t, dsn) + + testBetweenTablesSweep = func(queryer) { + loser.RunSweep(ctx) + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + winner.RunSweep(ctx) // deleting pass: office_routine_runs, pause, then runs + + winnerLast, ok := winner.LastSweepSnapshot() + if !ok { + t.Fatal("winner LastSweepSnapshot: ok = false, want true") + } + if winnerLast.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("winner office_routine_runs.Deleted = %d, want 1", winnerLast.OfficeRoutineRuns.Deleted) + } + if winnerLast.Runs.Deleted != 1 { + t.Fatalf("winner runs.Deleted = %d, want 1 (winner must complete both tables)", winnerLast.Runs.Deleted) + } + + if _, ok := loser.LastSweepSnapshot(); ok { + t.Fatal("loser LastSweepSnapshot: ok = true, want false (loser must not have run)") + } + loserSkips, _ := loser.SkipSnapshot() + if loserSkips != 1 { + t.Fatalf("loser skip count = %d, want 1", loserSkips) + } +} + +// TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire is +// the companion test the design's Testing section requires: dropping the +// winner's lock connection mid-sweep must stop it before the next table +// (F25) and record the whole attempt as a skip rather than a partial +// result (F28), never attempting to re-acquire. +func TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + victim, conn := newPostgresTestSweeper(t, dsn) + saveZeroFloorSettings(t, victim) + + // The pool's one connection (SetMaxOpenConns(1)) is what pg_terminate_backend + // kills below; database/sql transparently opens a replacement on the + // next query, which starts on the default search_path rather than + // this test's isolated schema. Capture the schema now to restore it + // before any post-mortem query on conn. + var schema string + if err := conn.Get(&schema, `SELECT current_schema()`); err != nil { + t.Fatalf("select current_schema: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + victim.RunSweep(ctx) // preview pass for both tables + + admin := testutil.OpenIsolatedPostgres(t, dsn) + adminPool := db.NewPool(admin, admin) + + testBetweenTablesSweep = func(q queryer) { + var pid int + if err := q.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil { + t.Fatalf("select pg_backend_pid: %v", err) + } + if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil { + t.Fatalf("terminate lock connection: %v", err) + } + // pg_terminate_backend signals the backend asynchronously; wait + // for it to actually leave pg_stat_activity before returning, so + // the alive() check right after this hook is not racing the + // signal's delivery. + deadline := time.Now().Add(5 * time.Second) + for { + var stillThere bool + if err := adminPool.Writer().GetContext(ctx, &stillThere, + `SELECT EXISTS(SELECT 1 FROM pg_stat_activity WHERE pid = $1)`, pid, + ); err != nil { + t.Fatalf("poll pg_stat_activity: %v", err) + } + if !stillThere { + return + } + if time.Now().After(deadline) { + t.Fatalf("backend %d still present in pg_stat_activity after 5s", pid) + } + time.Sleep(10 * time.Millisecond) + } + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + beforeAttempt, ok := victim.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot after the preview pass: ok = false, want true") + } + + victim.RunSweep(ctx) // deleting pass: office_routine_runs succeeds, then the lock connection dies + + // The preview pass already set LastSweep; a lock lost mid-attempt + // must leave it exactly as-is rather than publishing a partial + // result (F28) — not become unset, which would also be true after a + // genuinely successful sweep with nothing yet recorded. + afterAttempt, ok := victim.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot after the lost-lock attempt: ok = false, want true (still the preview pass's result)") + } + if afterAttempt != beforeAttempt { + t.Fatalf("LastSweep changed after a lost-lock attempt: before=%+v after=%+v", beforeAttempt, afterAttempt) + } + skips, _ := victim.SkipSnapshot() + if skips != 1 { + t.Fatalf("skip count = %d, want 1", skips) + } + + // conn's one physical connection was the one just terminated; + // database/sql opened a replacement on the default search_path, so + // restore the isolated schema before verifying table state. + if _, err := conn.Exec("SET search_path TO " + schema); err != nil { + t.Fatalf("restore search_path: %v", err) + } + + // office_routine_runs' delete committed before the session died + // (AC-002.5: batches already committed stay committed). + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 { + t.Fatalf("office_routine_runs rows = %d, want 0 (the first table's committed delete survives)", n) + } + // runs was never reached. + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 { + t.Fatalf("runs rows = %d, want 1 (the second table must not have been touched)", n) + } + + // A fresh session can now acquire the lock: it was not re-acquired by + // the victim, and PostgreSQL released it when the session ended. + replacement, ok, err := acquireSweepSession(ctx, adminPool) + if err != nil { + t.Fatalf("acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("ok = false, want true (the terminated session must have released the lock)") + } + replacement.release() +} diff --git a/apps/backend/internal/office/retention/sweep_test.go b/apps/backend/internal/office/retention/sweep_test.go new file mode 100644 index 00000000000..1f8de95995a --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_test.go @@ -0,0 +1,325 @@ +package retention + +import ( + "context" + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// newTestSweeper builds a Sweeper over one in-memory SQLite database +// carrying both the office schema (routine runs, plain runs, satellites) +// and the settings schema, so a sweep's table deletes and its +// settings/preview reads share one connection exactly as they do on the +// writer pool in production. +func newTestSweeper(t *testing.T) (*Sweeper, *sqlx.DB) { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + + store := NewStore(pool) + settingsStore := NewSettingsStore(settingsRaw) + previewMarker := NewPreviewMarkerStore(settingsRaw) + return NewSweeper(pool, store, settingsStore, previewMarker), conn +} + +func TestRunSweep_FirstPassPreviewsBothTablesWithoutDeleting(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if !last.OfficeRoutineRuns.Previewed || last.OfficeRoutineRuns.WouldDelete != 1 { + t.Fatalf("office_routine_runs = %+v, want previewed with WouldDelete=1", last.OfficeRoutineRuns) + } + if last.OfficeRoutineRuns.Deleted != 0 { + t.Fatalf("office_routine_runs.Deleted = %d, want 0 on a preview pass", last.OfficeRoutineRuns.Deleted) + } + if !last.Runs.Previewed || last.Runs.WouldDelete != 1 { + t.Fatalf("runs = %+v, want previewed with WouldDelete=1", last.Runs) + } + if last.Runs.Deleted != 0 { + t.Fatalf("runs.Deleted = %d, want 0 on a preview pass", last.Runs.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 { + t.Fatalf("office_routine_runs rows after preview = %d, want 1 (nothing deleted)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 { + t.Fatalf("runs rows after preview = %d, want 1 (nothing deleted)", n) + } +} + +func TestRunSweep_SecondPassDeletesAfterPreview(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if last.OfficeRoutineRuns.Previewed { + t.Fatal("office_routine_runs: second sweep should not be a preview pass") + } + if last.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1", last.OfficeRoutineRuns.Deleted) + } + if last.Runs.Previewed { + t.Fatal("runs: second sweep should not be a preview pass") + } + if last.Runs.Deleted != 1 { + t.Fatalf("runs.Deleted = %d, want 1", last.Runs.Deleted) + } + if last.RunEvents.Deleted != 1 { + t.Fatalf("run_events.Deleted = %d, want 1", last.RunEvents.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 { + t.Fatalf("office_routine_runs rows after delete = %d, want 0", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 { + t.Fatalf("runs rows after delete = %d, want 0", n) + } +} + +func TestRunSweep_BacklogFlaggedWhenEligibleExceedsBatchLimit(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 // minBatchLimit; AC-004.3 forbids going lower + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass, no deletion, no batch limit involved + sweeper.RunSweep(ctx) // deleting pass: 105 eligible, batch limit 100 + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false") + } + if last.OfficeRoutineRuns.Deleted != batchLimit { + t.Fatalf("Deleted = %d, want %d (capped by batch limit)", last.OfficeRoutineRuns.Deleted, batchLimit) + } + if !last.OfficeRoutineRuns.Backlog { + t.Fatal("Backlog = false, want true (105 eligible > batch limit 100)") + } +} + +func TestRunSweep_DisabledSkipsSweepWithoutRecordingSkip(t *testing.T) { + sweeper, _ := newTestSweeper(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false (disabled means no sweep ran)") + } + count, _ := sweeper.SkipSnapshot() + if count != 0 { + t.Fatalf("skip count = %d, want 0 (disabled is not a recorded skip)", count) + } +} + +func TestRunSweep_SettingsUnreadableSkipsAndRecordsSkip(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false") + } + count, lastAt := sweeper.SkipSnapshot() + if count != 1 { + t.Fatalf("skip count = %d, want 1", count) + } + if lastAt.IsZero() { + t.Fatal("lastAt is zero, want a recorded skip time") + } +} + +func TestRunSweep_ConcurrentAttemptRecordsSkipWithoutRunning(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + + // Simulate a sweep already in flight rather than racing goroutines + // against SQLite's speed, which would make the collision + // non-deterministic. + sweeper.mu.Lock() + sweeper.sweeping = true + sweeper.mu.Unlock() + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false (the attempt must not have run)") + } + count, _ := sweeper.SkipSnapshot() + if count != 1 { + t.Fatalf("skip count = %d, want 1", count) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 { + t.Fatalf("rows = %d, want 1 (a blocked attempt must not preview or delete)", n) + } +} + +func TestRunSweep_AbandonedRunsBatchReportsFailureNotBacklogWithZeroSatellites(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + old := daysAgo(60) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + + // Before each attempt's selection, make sure the row reads terminal + // again (undoing the previous attempt's resurrection) so it is + // selected every time; after selection, flip it live so that + // attempt's own delete re-assertion mismatches — forcing both + // attempts to roll back and the batch to abandon. + testBeforeSelectEligibleRunIDs = func(attempt int) { + if attempt == 0 { + return // already terminal from seeding + } + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID) + } + testAfterSelectEligibleRunIDs = func(int, []string) { + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID) + } + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + + sweeper.RunSweep(ctx) // deleting pass: forced to abandon + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false") + } + if last.Runs.Err == "" { + t.Fatal("runs.Err is empty, want the abandon failure recorded") + } + if last.Runs.Backlog { + t.Fatal("runs.Backlog = true, want false (an abandoned batch is a failure, not backlog)") + } + if last.Runs.Deleted != 0 { + t.Fatalf("runs.Deleted = %d, want 0", last.Runs.Deleted) + } + if last.RunEvents.Deleted != 0 { + t.Fatalf("run_events.Deleted = %d, want 0 (rollback restored it)", last.RunEvents.Deleted) + } +} + +func TestLastSweepSnapshot_FalseBeforeFirstSweep(t *testing.T) { + sweeper, _ := newTestSweeper(t) + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("ok = true before any sweep has run, want false") + } +} + +func TestRunCensus_PopulatesRetainedCountsForAllThreeTables(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunCensus(ctx) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh || counts.OfficeRoutineRuns.RetainedCount != 1 { + t.Fatalf("office_routine_runs census = %+v, want fresh with 1", counts.OfficeRoutineRuns) + } + if counts.Runs.State != CensusFresh || counts.Runs.RetainedCount != 1 { + t.Fatalf("runs census = %+v, want fresh with 1", counts.Runs) + } + if counts.RunEvents.State != CensusFresh || counts.RunEvents.RetainedCount != 1 { + t.Fatalf("run_events census = %+v, want fresh with 1", counts.RunEvents) + } +} + +// saveZeroFloorSettings drops both tables' floor to 0 so a test's single +// seeded row is eligible: the default floor of 50 protects the newest 50 +// rows per owner, which a one-row fixture never exceeds. +func saveZeroFloorSettings(t *testing.T, sweeper *Sweeper) { + t.Helper() + settings := DefaultSettings() + settings.RoutineRuns.FloorPerOwner = 0 + settings.Runs.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } +} diff --git a/apps/backend/internal/office/retention/types.go b/apps/backend/internal/office/retention/types.go new file mode 100644 index 00000000000..63e7451353b --- /dev/null +++ b/apps/backend/internal/office/retention/types.go @@ -0,0 +1,146 @@ +// Package retention bounds Office run history: office_routine_runs, runs, +// and the run satellite tables (run_events, office_run_route_attempts, +// office_run_skills). See docs/specs/office/requirements/run-history-retention.md +// and its operations counterpart for the frozen contract this package +// implements. +package retention + +import ( + "errors" + "fmt" +) + +// ErrValidation is returned by NormalizeSettings when a field is outside its +// permitted range. The error message names the field. +var ErrValidation = errors.New("retention settings validation") + +// ErrInvalidPersistedSettings wraps a stored settings document that could +// not be read or parsed. Callers fall back to DefaultSettings and surface a +// health issue rather than failing. +var ErrInvalidPersistedSettings = errors.New("invalid persisted retention settings") + +// TableName identifies one of the tables retention reasons about. +type TableName string + +const ( + TableOfficeRoutineRuns TableName = "office_routine_runs" + TableRuns TableName = "runs" + TableRunEvents TableName = "run_events" +) + +// SweptTables are the two tables retention selects rows from by policy, in +// the fixed sweep order. +var SweptTables = []TableName{TableOfficeRoutineRuns, TableRuns} + +// ReportedTables are the swept tables plus the three run satellites that a +// sweep can delete rows from. +var ReportedTables = []TableName{ + TableOfficeRoutineRuns, TableRuns, + "run_events", "office_run_route_attempts", "office_run_skills", +} + +// ThresholdedTables carry a warning threshold on retained row count. +var ThresholdedTables = []TableName{TableOfficeRoutineRuns, TableRuns, TableRunEvents} + +// TableSettings is the per-table retention policy for a swept table. +type TableSettings struct { + WindowDays int `json:"window_days"` + FloorPerOwner int `json:"floor_per_owner"` + WarnRows int `json:"warn_rows"` +} + +// RunEventsSettings is the threshold-only policy for run_events, which has +// no window or floor of its own: its lifetime is its run's. +type RunEventsSettings struct { + WarnRows int `json:"warn_rows"` +} + +// Settings is the full retention policy document, persisted under one +// settings-store key as JSON. +type Settings struct { + Enabled bool `json:"enabled"` + SweepIntervalHours int `json:"sweep_interval_hours"` + BatchLimit int `json:"batch_limit"` + RoutineRuns TableSettings `json:"routine_runs"` + Runs TableSettings `json:"runs"` + RunEvents RunEventsSettings `json:"run_events"` +} + +// DefaultSettings returns the documented AC-OFFICE-RUN-HISTORY-RETENTION-004.2 +// defaults. +func DefaultSettings() Settings { + return Settings{ + Enabled: true, + SweepIntervalHours: 6, + BatchLimit: 5000, + RoutineRuns: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}, + Runs: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}, + RunEvents: RunEventsSettings{WarnRows: 250000}, + } +} + +// Permitted ranges, AC-OFFICE-RUN-HISTORY-RETENTION-004.3. +const ( + minWindowDays = 1 + maxWindowDays = 3650 + + minSweepIntervalHours = 1 + maxSweepIntervalHours = 168 + + minFloorPerOwner = 0 + maxFloorPerOwner = 10000 + + minBatchLimit = 100 + maxBatchLimit = 100000 + + minWarnRows = 0 +) + +// NormalizeSettings validates every field against its permitted range and +// returns a field-named error on the first violation, changing nothing. +// Ranges are inclusive on both ends. +func NormalizeSettings(in Settings) (Settings, error) { + if err := validateRange("sweep_interval_hours", in.SweepIntervalHours, minSweepIntervalHours, maxSweepIntervalHours); err != nil { + return Settings{}, err + } + if err := validateRange("batch_limit", in.BatchLimit, minBatchLimit, maxBatchLimit); err != nil { + return Settings{}, err + } + if err := validateTableSettings("routine_runs", in.RoutineRuns); err != nil { + return Settings{}, err + } + if err := validateTableSettings("runs", in.Runs); err != nil { + return Settings{}, err + } + if err := validateMin("run_events.warn_rows", in.RunEvents.WarnRows, minWarnRows); err != nil { + return Settings{}, err + } + return in, nil +} + +func validateTableSettings(prefix string, s TableSettings) error { + if err := validateRange(prefix+".window_days", s.WindowDays, minWindowDays, maxWindowDays); err != nil { + return err + } + if err := validateRange(prefix+".floor_per_owner", s.FloorPerOwner, minFloorPerOwner, maxFloorPerOwner); err != nil { + return err + } + if err := validateMin(prefix+".warn_rows", s.WarnRows, minWarnRows); err != nil { + return err + } + return nil +} + +func validateRange(field string, value, minValue, maxValue int) error { + if value < minValue || value > maxValue { + return fmt.Errorf("%w: %s must be between %d and %d", ErrValidation, field, minValue, maxValue) + } + return nil +} + +func validateMin(field string, value, minValue int) error { + if value < minValue { + return fmt.Errorf("%w: %s must be %d or greater", ErrValidation, field, minValue) + } + return nil +} diff --git a/apps/backend/internal/office/retention/types_test.go b/apps/backend/internal/office/retention/types_test.go new file mode 100644 index 00000000000..00b61d94d14 --- /dev/null +++ b/apps/backend/internal/office/retention/types_test.go @@ -0,0 +1,94 @@ +package retention + +import "testing" + +func TestDefaultSettings_MatchesDocumentedDefaults(t *testing.T) { + got := DefaultSettings() + + if !got.Enabled { + t.Fatalf("Enabled = false, want true") + } + if got.SweepIntervalHours != 6 { + t.Fatalf("SweepIntervalHours = %d, want 6", got.SweepIntervalHours) + } + if got.BatchLimit != 5000 { + t.Fatalf("BatchLimit = %d, want 5000", got.BatchLimit) + } + wantRoutineRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000} + if got.RoutineRuns != wantRoutineRuns { + t.Fatalf("RoutineRuns = %+v, want %+v", got.RoutineRuns, wantRoutineRuns) + } + wantRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000} + if got.Runs != wantRuns { + t.Fatalf("Runs = %+v, want %+v", got.Runs, wantRuns) + } + if got.RunEvents.WarnRows != 250000 { + t.Fatalf("RunEvents.WarnRows = %d, want 250000", got.RunEvents.WarnRows) + } +} + +func TestNormalizeSettings_AcceptsDefaults(t *testing.T) { + normalized, err := NormalizeSettings(DefaultSettings()) + if err != nil { + t.Fatalf("NormalizeSettings(defaults): %v", err) + } + if normalized != DefaultSettings() { + t.Fatalf("NormalizeSettings(defaults) = %+v, want unchanged defaults", normalized) + } +} + +func TestNormalizeSettings_RejectsOutOfRangeFields(t *testing.T) { + cases := []struct { + name string + mutate func(*Settings) + wantErr string + }{ + {"window too low", func(s *Settings) { s.RoutineRuns.WindowDays = 0 }, "routine_runs.window_days"}, + {"window too high", func(s *Settings) { s.Runs.WindowDays = 3651 }, "runs.window_days"}, + {"interval too low", func(s *Settings) { s.SweepIntervalHours = 0 }, "sweep_interval_hours"}, + {"interval too high", func(s *Settings) { s.SweepIntervalHours = 169 }, "sweep_interval_hours"}, + {"floor negative", func(s *Settings) { s.RoutineRuns.FloorPerOwner = -1 }, "routine_runs.floor_per_owner"}, + {"floor too high", func(s *Settings) { s.Runs.FloorPerOwner = 10001 }, "runs.floor_per_owner"}, + {"batch too low", func(s *Settings) { s.BatchLimit = 99 }, "batch_limit"}, + {"batch too high", func(s *Settings) { s.BatchLimit = 100001 }, "batch_limit"}, + {"warn negative", func(s *Settings) { s.RoutineRuns.WarnRows = -1 }, "routine_runs.warn_rows"}, + {"run_events warn negative", func(s *Settings) { s.RunEvents.WarnRows = -1 }, "run_events.warn_rows"}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + s := DefaultSettings() + tc.mutate(&s) + _, err := NormalizeSettings(s) + if err == nil { + t.Fatalf("NormalizeSettings(%+v): want error, got nil", s) + } + if got := err.Error(); !containsField(got, tc.wantErr) { + t.Fatalf("NormalizeSettings error = %q, want it to name field %q", got, tc.wantErr) + } + }) + } +} + +func TestNormalizeSettings_ZeroWarnRowsDisablesThreshold(t *testing.T) { + s := DefaultSettings() + s.RoutineRuns.WarnRows = 0 + s.Runs.WarnRows = 0 + s.RunEvents.WarnRows = 0 + if _, err := NormalizeSettings(s); err != nil { + t.Fatalf("NormalizeSettings with zero warn thresholds: %v", err) + } +} + +func containsField(msg, field string) bool { + return len(msg) >= len(field) && (indexOf(msg, field) >= 0) +} + +func indexOf(haystack, needle string) int { + for i := 0; i+len(needle) <= len(haystack); i++ { + if haystack[i:i+len(needle)] == needle { + return i + } + } + return -1 +} diff --git a/apps/backend/internal/office/service/scheduler_checkout_error_test.go b/apps/backend/internal/office/service/scheduler_checkout_error_test.go index 1cd259ec9ee..05f250781ac 100644 --- a/apps/backend/internal/office/service/scheduler_checkout_error_test.go +++ b/apps/backend/internal/office/service/scheduler_checkout_error_test.go @@ -120,8 +120,17 @@ func TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails(t *testing. // the tasks table, and every other runs column, writable — a targeted // fault instead of a global read-only pragma, so this test actually // distinguishes "release before finish" from "finish before release" - // rather than failing both writes identically. - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + // rather than failing both writes identically. A trigger rather than + // DROP COLUMN: idx_runs_retention is an expression index over + // COALESCE(finished_at, ...), and SQLite refuses to drop a column an + // index still references. + svc.ExecSQL(t, ` + CREATE TRIGGER block_finish_order_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) event := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "task_id": "task-finish-order-1", diff --git a/apps/web/components/settings/system/data-logs-settings.tsx b/apps/web/components/settings/system/data-logs-settings.tsx index 93ff76a4c46..5ef25f4601a 100644 --- a/apps/web/components/settings/system/data-logs-settings.tsx +++ b/apps/web/components/settings/system/data-logs-settings.tsx @@ -7,6 +7,7 @@ import { SettingsTarget } from "@/components/settings/settings-target"; import { BackupsTable } from "@/components/settings/system/backups-table"; import { DatabaseStatsCard } from "@/components/settings/system/database-stats-card"; import { LogViewer } from "@/components/settings/system/log-viewer"; +import { RetentionSettingsCard } from "@/components/settings/system/retention-settings-card"; import { BACKUP_SQL_COMMAND } from "@/components/settings/system/system-route-shell"; import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system"; @@ -35,6 +36,14 @@ export function DataLogsSettings() { + + + + + ({ + fetchRetentionStatus: (...args: unknown[]) => fetchRetentionStatusMock(...args), + saveRetentionSettings: (...args: unknown[]) => saveRetentionSettingsMock(...args), +})); + +vi.mock("@/components/settings/settings-save-provider", () => ({ + useSettingsSaveContributor: (contributor: SettingsSaveContributor) => { + saveContributor = contributor; + }, +})); + +import { RetentionSettingsCard } from "./retention-settings-card"; + +function defaultSettings(overrides: Partial = {}): RetentionSettings { + return { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + ...overrides, + }; +} + +function statusOf(overrides: Partial = {}): RetentionStatus { + return { + settings: defaultSettings(), + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + ...overrides, + }; +} + +function renderCard() { + return render( + + + , + ); +} + +beforeEach(() => { + fetchRetentionStatusMock.mockReset(); + saveRetentionSettingsMock.mockReset(); + fetchRetentionStatusMock.mockResolvedValue(statusOf()); + currentRole = "admin"; + saveContributor = null; +}); + +afterEach(() => { + cleanup(); + vi.clearAllMocks(); +}); + +describe("RetentionSettingsCard", () => { + it("loads settings and renders the swept-table fields from the fetched status", async () => { + renderCard(); + + const windowDays = await screen.findByTestId("retention-routine-runs-window-days"); + expect(windowDays).toHaveProperty("value", "30"); + expect(screen.getByTestId("retention-runs-warn-rows")).toHaveProperty("value", "25000"); + expect(screen.getByTestId("retention-run-events-warn-rows")).toHaveProperty("value", "250000"); + expect(screen.getByTestId("retention-never-swept")).toBeTruthy(); + }); + + it("stages an admin edit until the shared save contributor runs, then reloads", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + + const toggle = screen.getByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(toggle); + expect(saveRetentionSettingsMock).not.toHaveBeenCalled(); + expect(saveContributor?.isDirty).toBe(true); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); + fetchRetentionStatusMock.mockResolvedValueOnce( + statusOf({ settings: defaultSettings({ enabled: false }) }), + ); + + await act(async () => saveContributor?.save(saveContributor.revision)); + + expect(saveRetentionSettingsMock).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + ); + await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); + }); + + it("keeps members read-only while preserving the loaded values", async () => { + currentRole = "member"; + renderCard(); + + const windowDays = await screen.findByTestId("retention-routine-runs-window-days"); + expect(windowDays).toHaveProperty("disabled", true); + expect(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)).toHaveProperty("disabled", true); + expect(screen.getByText("Only an admin can change retention settings.")).toBeTruthy(); + expect(saveContributor?.isDirty).toBe(false); + }); + + it("reports a failed save without clearing the dirty draft", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockRejectedValueOnce(new Error("offline")); + await act(async () => { + await expect(saveContributor?.save(saveContributor.revision)).rejects.toThrow("offline"); + }); + + expect(saveContributor?.isDirty).toBe(true); + await screen.findByTestId("retention-save-error"); + }); + + it("renders the last sweep outcome, backlog flag, and retained counts", async () => { + fetchRetentionStatusMock.mockResolvedValue( + statusOf({ + last_sweep: { + started_at: "2026-09-01T00:00:00Z", + finished_at: "2026-09-01T00:00:05Z", + office_routine_runs: { + deleted: 12, + backlog: true, + error: "", + previewed: false, + would_delete: 0, + }, + runs: { deleted: 3, backlog: false, error: "", previewed: true, would_delete: 40 }, + run_events: { deleted: 100, backlog: false, error: "" }, + route_attempts: { deleted: 0, backlog: false, error: "" }, + run_skills: { deleted: 0, backlog: false, error: "" }, + }, + skip_count: 2, + last_skip_at: "2026-09-01T00:10:00Z", + retained_counts: { + office_routine_runs: { + state: "fresh", + retained_count: 1200, + as_of: "2026-09-01T00:00:00Z", + top_routine_id: "routine-1", + top_routine_share: 0.42, + }, + runs: { state: "stale", retained_count: 800, as_of: "2026-08-31T00:00:00Z" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + renderCard(); + + await screen.findByTestId("retention-last-sweep"); + expect(screen.getByTestId("retention-backlog-office_routine_runs")).toBeTruthy(); + expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain( + "1200", + ); + expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain( + "42%", + ); + expect(screen.getByText(/Stale: last measurement failed/)).toBeTruthy(); + expect(screen.getByTestId("retention-skip-count").textContent).toContain("2"); + }); +}); diff --git a/apps/web/components/settings/system/retention-settings-card.tsx b/apps/web/components/settings/system/retention-settings-card.tsx new file mode 100644 index 00000000000..282979a07f2 --- /dev/null +++ b/apps/web/components/settings/system/retention-settings-card.tsx @@ -0,0 +1,551 @@ +"use client"; + +import { useEffect, useRef, useState, type ReactNode } from "react"; +import { useTranslation } from "react-i18next"; +import { Alert, AlertDescription } from "@kandev/ui/alert"; +import { CardContent } from "@kandev/ui/card"; +import { Input } from "@kandev/ui/input"; +import { Spinner } from "@kandev/ui/spinner"; +import { Switch } from "@kandev/ui/switch"; +import { IconAlertCircle } from "@tabler/icons-react"; +import { SettingsCard } from "@/components/settings/settings-card"; +import { SettingsCardHeader } from "@/components/settings/settings-card-header"; +import { settingsControlClassName } from "@/components/settings/settings-control"; +import { + SettingsFieldDescription, + SettingsFieldLabel, +} from "@/components/settings/settings-typography"; +import { useSettingsSaveContributor } from "@/components/settings/settings-save-provider"; +import { useIsAdmin } from "@/hooks/domains/auth/use-is-admin"; +import { useRetentionSettings } from "@/hooks/domains/system/use-retention-settings"; +import { formatDateTime } from "@/lib/i18n/formats"; +import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system"; +import type { + RetentionSettings, + RetentionStatus, + RetentionTableCensus, + RetentionTableSweepResult, + RetentionSweptTableResult, +} from "@/lib/types/system"; + +function serialize(settings: RetentionSettings | null): string { + return settings ? JSON.stringify(settings) : "loading"; +} + +function NumberField({ + label, + help, + value, + min, + max, + disabled, + onChange, + testId, +}: { + label: string; + help: string; + value: number; + min: number; + max?: number; + disabled?: boolean; + onChange: (value: number) => void; + testId: string; +}) { + return ( +
+ {label} + onChange(Number(event.target.value))} + className={settingsControlClassName("h-11")} + data-testid={testId} + /> + {help} +
+ ); +} + +function useRetentionDraft(remote: ReturnType, isAdmin: boolean) { + const { t } = useTranslation(); + const [draft, setDraft] = useState(null); + const previousSaved = useRef(null); + const saved = remote.status?.settings ?? null; + + useEffect(() => { + if (!saved) return; + setDraft((current) => { + const previous = previousSaved.current; + if (!current || !previous || serialize(current) === serialize(previous)) return saved; + return current; + }); + previousSaved.current = saved; + }, [saved]); + + const isDirty = Boolean(draft && saved && serialize(draft) !== serialize(saved)); + const canEdit = isAdmin && !remote.isLoading && Boolean(saved); + const invalidReason = !isAdmin ? t("system:retentionAdminOnly") : undefined; + + useSettingsSaveContributor({ + id: "system:retention", + order: 25, + revision: serialize(draft), + isDirty, + canSave: canEdit, + invalidReason, + save: async () => { + if (!draft) return; + await remote.save(draft); + }, + discard: () => { + if (saved) setDraft(saved); + }, + }); + + return { draft, setDraft, saved, canEdit }; +} + +function RetentionEnabledRow({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( +
+
+ + {t("system:retentionEnabledLabel")} + + + {t("system:retentionEnabledDescription")} + +
+ onChange({ ...settings, enabled })} + data-testid="retention-enabled" + className="shrink-0 cursor-pointer" + /> +
+ ); +} + +function RetentionScheduleFields({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( +
+ onChange({ ...settings, sweep_interval_hours })} + testId="retention-sweep-interval" + /> + onChange({ ...settings, batch_limit })} + testId="retention-batch-limit" + /> +
+ ); +} + +function TableSection({ + title, + description, + children, +}: { + title: string; + description: string; + children: ReactNode; +}) { + return ( +
+
+

{title}

+

{description}

+
+
{children}
+
+ ); +} + +type WindowedTableKey = "routine_runs" | "runs"; + +function WindowedTableSection({ + tableKey, + title, + description, + settings, + disabled, + onChange, +}: { + tableKey: WindowedTableKey; + title: string; + description: string; + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + const table = settings[tableKey]; + const testPrefix = tableKey === "routine_runs" ? "retention-routine-runs" : "retention-runs"; + return ( + + onChange({ ...settings, [tableKey]: { ...table, window_days } })} + testId={`${testPrefix}-window-days`} + /> + + onChange({ ...settings, [tableKey]: { ...table, floor_per_owner } }) + } + testId={`${testPrefix}-floor-per-owner`} + /> + onChange({ ...settings, [tableKey]: { ...table, warn_rows } })} + testId={`${testPrefix}-warn-rows`} + /> + + ); +} + +function RunEventsSection({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( + + onChange({ ...settings, run_events: { warn_rows } })} + testId="retention-run-events-warn-rows" + /> + + ); +} + +function RoutineRunsAndRunsSections({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( + <> + + + + + ); +} + +function RetentionPolicyCard({ + draft, + canEdit, + onChange, +}: { + draft: RetentionSettings; + canEdit: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + const disabled = !canEdit; + return ( + + + + + + + {!canEdit && ( +

{t("system:retentionAdminOnly")}

+ )} +
+
+ ); +} + +function SweptTableRow({ label, result }: { label: string; result: RetentionSweptTableResult }) { + const { t } = useTranslation(); + return ( +
+ {label} + + {t("system:retentionDeletedLabel")}: {result.deleted} + + {result.previewed && ( + + {t("system:retentionWouldDeleteLabel")}: {result.would_delete} + + )} + {result.backlog && ( + + {t("system:retentionBacklogLabel")} + + )} + {result.error && ( + + {t("system:retentionTableErrorLabel")}: {result.error} + + )} +
+ ); +} + +function SatelliteTableRow({ + label, + result, +}: { + label: string; + result: RetentionTableSweepResult; +}) { + const { t } = useTranslation(); + return ( +
+ {label} + + {t("system:retentionDeletedLabel")}: {result.deleted} + + {result.error && ( + + {t("system:retentionTableErrorLabel")}: {result.error} + + )} +
+ ); +} + +function LastSweepSection({ status }: { status: RetentionStatus }) { + const { t } = useTranslation(); + const lastSweep = status.last_sweep; + if (!lastSweep) { + return ( +

+ {t("system:retentionNeverSweptMessage")} +

+ ); + } + return ( +
+

+ {t("system:retentionSweepStartedAtLabel")}: {formatDateTime(lastSweep.started_at)} + {" · "} + {t("system:retentionSweepFinishedAtLabel")}: {formatDateTime(lastSweep.finished_at)} +

+ + + + + +
+ ); +} + +function RetainedCountRow({ label, census }: { label: string; census: RetentionTableCensus }) { + const { t } = useTranslation(); + if (census.state === "not_computed") { + return ( +
+ {label} + {t("system:retentionCensusNotComputed")} +
+ ); + } + return ( +
+ {label} + {census.retained_count} + + {t("system:retentionCensusAsOfLabel")}: {formatDateTime(census.as_of)} + + {census.state === "stale" && ( + {t("system:retentionCensusStale")} + )} + {census.unknown_statuses && census.unknown_statuses.length > 0 && ( + + {t("system:retentionUnknownStatusesLabel")}: {census.unknown_statuses.join(", ")} + + )} + {census.top_routine_id && ( + + {t("system:retentionTopRoutineShareLabel")}:{" "} + {Math.round((census.top_routine_share ?? 0) * 100)}% ({census.top_routine_id}) + + )} +
+ ); +} + +function RetentionStatusCard({ status }: { status: RetentionStatus | null }) { + const { t } = useTranslation(); + if (!status) return null; + return ( + + + + +
+

{t("system:retentionRetainedCountsTitle")}

+ + + +
+ {status.skip_count > 0 && ( +

+ {t("system:retentionSkipCountLabel")}: {status.skip_count} + {status.last_skip_at && ( + <> + {" · "} + {t("system:retentionLastSkipAtLabel")}: {formatDateTime(status.last_skip_at)} + + )} +

+ )} +
+
+ ); +} + +function RetentionSettingsLoading() { + const { t } = useTranslation(); + return ( + + + + {t("settings:loading")} + + + ); +} + +function RetentionSettingsLoadError({ error }: { error: string }) { + const { t } = useTranslation(); + return ( + + + + {t("system:retentionLoadFailed")}: {error} + + + ); +} + +export function RetentionSettingsCard() { + const remote = useRetentionSettings(); + const isAdmin = useIsAdmin(); + const { draft, setDraft, canEdit } = useRetentionDraft(remote, isAdmin); + const { t } = useTranslation(); + + if (remote.isLoading && !remote.status) return ; + if (remote.error && !remote.status) return ; + + return ( +
+ {draft && } + + {remote.saveError && ( + + + + {t("system:retentionSaveFailed")}: {remote.saveError} + + + )} +
+ ); +} diff --git a/apps/web/hooks/domains/system/use-retention-settings.ts b/apps/web/hooks/domains/system/use-retention-settings.ts new file mode 100644 index 00000000000..18dd4d65ae4 --- /dev/null +++ b/apps/web/hooks/domains/system/use-retention-settings.ts @@ -0,0 +1,49 @@ +"use client"; + +import { useCallback, useEffect, useState } from "react"; +import { useAppStore } from "@/components/state-provider"; +import { fetchRetentionStatus, saveRetentionSettings } from "@/lib/api/domains/system-api"; +import type { RetentionSettings } from "@/lib/types/system"; + +export function useRetentionSettings() { + const status = useAppStore((s) => s.system.retention); + const setStatus = useAppStore((s) => s.setSystemRetention); + const [isLoading, setIsLoading] = useState(false); + const [error, setError] = useState(null); + const [saveError, setSaveError] = useState(null); + + const reload = useCallback(async () => { + setIsLoading(true); + setError(null); + try { + setStatus(await fetchRetentionStatus({ cache: "no-store" })); + } catch (e) { + setError(e instanceof Error ? e.message : String(e)); + } finally { + setIsLoading(false); + } + }, [setStatus]); + + useEffect(() => { + if (status) return; + void reload(); + }, [status, reload]); + + const save = useCallback( + async (settings: RetentionSettings) => { + setSaveError(null); + try { + const saved = await saveRetentionSettings(settings); + await reload(); + return saved; + } catch (e) { + const message = e instanceof Error ? e.message : String(e); + setSaveError(message); + throw e; + } + }, + [reload], + ); + + return { status, isLoading, error, saveError, reload, save }; +} diff --git a/apps/web/lib/api/domains/system-api.test.ts b/apps/web/lib/api/domains/system-api.test.ts index c1980b1dd63..fda9156ac2c 100644 --- a/apps/web/lib/api/domains/system-api.test.ts +++ b/apps/web/lib/api/domains/system-api.test.ts @@ -43,6 +43,8 @@ import { restoreStorageQuarantine, runStorageMaintenance, saveStorageSettings, + fetchRetentionStatus, + saveRetentionSettings, } from "./system-api"; const BASE = "http://api.test/api/v1/system"; @@ -528,3 +530,47 @@ describe("storage policy", () => { expect(response.capabilities.docker_available).toBe(true); }); }); + +describe("office run history retention", () => { + const retentionSettings = { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + }; + + it("loads retention status without caching", async () => { + fetchSpy.mockResolvedValueOnce( + jsonResponse({ + settings: retentionSettings, + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + + const response = await fetchRetentionStatus(); + + expect(lastCall().url).toBe(`${BASE}/retention`); + expect(lastCall().init?.cache).toBe("no-store"); + expect(response.settings).toEqual(retentionSettings); + expect(response.last_sweep).toBeNull(); + }); + + it("PUTs the full settings document to save", async () => { + fetchSpy.mockResolvedValueOnce(jsonResponse(retentionSettings)); + + const response = await saveRetentionSettings(retentionSettings); + + expect(lastCall().url).toBe(`${BASE}/retention`); + expect(method()).toBe("PUT"); + expect(JSON.parse(String(lastCall().init?.body))).toEqual(retentionSettings); + expect(response).toEqual(retentionSettings); + }); +}); diff --git a/apps/web/lib/api/domains/system-api.ts b/apps/web/lib/api/domains/system-api.ts index 8c27004e7b8..4c5395cfd27 100644 --- a/apps/web/lib/api/domains/system-api.ts +++ b/apps/web/lib/api/domains/system-api.ts @@ -24,6 +24,8 @@ import type { StorageQuarantinePurgeScope, StorageSettingsResponse, UpdatesChannel, + RetentionSettings, + RetentionStatus, } from "@/lib/types/system"; const SYSTEM_BASE = "/api/v1/system"; @@ -416,3 +418,26 @@ export function purgeStorageQuarantine( }, }); } + +// --- Office run history retention ---------------------------------------- + +export function fetchRetentionStatus(options?: ApiRequestOptions): Promise { + return fetchJson(`${SYSTEM_BASE}/retention`, { + ...options, + cache: "no-store", + }); +} + +export function saveRetentionSettings( + settings: RetentionSettings, + options?: ApiRequestOptions, +): Promise { + return fetchJson(`${SYSTEM_BASE}/retention`, { + ...options, + init: { + ...(options?.init ?? {}), + method: "PUT", + body: JSON.stringify(settings), + }, + }); +} diff --git a/apps/web/lib/settings-discovery/catalog/system.ts b/apps/web/lib/settings-discovery/catalog/system.ts index f2f71da4bcc..82405c65736 100644 --- a/apps/web/lib/settings-discovery/catalog/system.ts +++ b/apps/web/lib/settings-discovery/catalog/system.ts @@ -9,6 +9,7 @@ export const SYSTEM_STORAGE_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/storage`; export const SYSTEM_ABOUT_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/about`; export const SYSTEM_SETTINGS_TARGETS = { database: "setting-system-database", + retention: "setting-system-retention", backups: "setting-system-backups", logs: "setting-system-logs", licenses: "setting-system-licenses", @@ -59,6 +60,16 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ targetId: SYSTEM_SETTINGS_TARGETS.database, order: 621, }, + { + id: "system-retention", + kind: "section", + labelKey: "system:navRetention", + parentId: SYSTEM_DATA_STORAGE_DISCOVERY_ID, + groupId: "system", + href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, + targetId: SYSTEM_SETTINGS_TARGETS.retention, + order: 622, + }, { id: "system-backups", kind: "section", @@ -67,7 +78,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ groupId: "system", href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, targetId: SYSTEM_SETTINGS_TARGETS.backups, - order: 622, + order: 623, }, { id: "system-logs", @@ -77,7 +88,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ groupId: "system", href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, targetId: SYSTEM_SETTINGS_TARGETS.logs, - order: 623, + order: 624, }, { id: "system-storage", diff --git a/apps/web/lib/state/slices/system/system-slice.test.ts b/apps/web/lib/state/slices/system/system-slice.test.ts index ef24efea1e4..00e1103e73e 100644 --- a/apps/web/lib/state/slices/system/system-slice.test.ts +++ b/apps/web/lib/state/slices/system/system-slice.test.ts @@ -200,6 +200,30 @@ describe("system slice", () => { expect(store.getState().system.database).toEqual(DB_STATS); }); + it("setSystemRetention stores the status", () => { + const store = makeStore(); + expect(store.getState().system.retention).toBeNull(); + const status = { + settings: { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + }, + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + runs: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + run_events: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + }, + }; + store.getState().setSystemRetention(status); + expect(store.getState().system.retention).toEqual(status); + }); + it("setSystemBackups marks the list as loaded", () => { const store = makeStore(); store.getState().setSystemBackups([SNAPSHOT]); diff --git a/apps/web/lib/state/slices/system/system-slice.ts b/apps/web/lib/state/slices/system/system-slice.ts index 22ed5baeaa8..fc26162b758 100644 --- a/apps/web/lib/state/slices/system/system-slice.ts +++ b/apps/web/lib/state/slices/system/system-slice.ts @@ -6,6 +6,7 @@ export const defaultSystemState: SystemSliceState = { info: null, diskUsage: null, database: null, + retention: null, backups: { items: [], loaded: false }, updates: null, jobs: {}, @@ -44,6 +45,10 @@ export const createSystemSlice: StateCreator< set((draft) => { draft.system.database = stats; }), + setSystemRetention: (status) => + set((draft) => { + draft.system.retention = status; + }), setSystemBackups: (items) => set((draft) => { draft.system.backups = { items, loaded: true }; diff --git a/apps/web/lib/state/slices/system/types.ts b/apps/web/lib/state/slices/system/types.ts index 7135fdaf363..ecb21316e54 100644 --- a/apps/web/lib/state/slices/system/types.ts +++ b/apps/web/lib/state/slices/system/types.ts @@ -11,6 +11,7 @@ import type { StorageOverviewResponse, StoragePolicyResponse, StorageQuarantineEntry, + RetentionStatus, } from "@/lib/types/system"; export type SystemBackupsState = { @@ -25,6 +26,7 @@ export type SystemSliceState = { info: SystemInfo | null; diskUsage: DiskUsageResponse | null; database: DatabaseStats | null; + retention: RetentionStatus | null; backups: SystemBackupsState; updates: UpdatesResponse | null; jobs: SystemJobsMap; @@ -44,6 +46,7 @@ export type SystemSliceActions = { setSystemInfo: (info: SystemInfo) => void; setSystemDiskUsage: (usage: DiskUsageResponse) => void; setSystemDatabase: (stats: DatabaseStats) => void; + setSystemRetention: (status: RetentionStatus) => void; setSystemBackups: (items: SnapshotInfo[]) => void; setSystemUpdates: (updates: UpdatesResponse) => void; upsertSystemJob: (job: SystemJob) => void; diff --git a/apps/web/lib/types/system.ts b/apps/web/lib/types/system.ts index 09949372fbb..fad1ad52136 100644 --- a/apps/web/lib/types/system.ts +++ b/apps/web/lib/types/system.ts @@ -560,6 +560,73 @@ export interface StorageAdoptionResponse extends StorageSettingsResponse { capabilities: StorageCapabilities; } +// --- Office run history retention --------------------------------------- + +export interface RetentionTableSettings { + window_days: number; + floor_per_owner: number; + warn_rows: number; +} + +export interface RetentionRunEventsSettings { + warn_rows: number; +} + +export interface RetentionSettings { + enabled: boolean; + sweep_interval_hours: number; + batch_limit: number; + routine_runs: RetentionTableSettings; + runs: RetentionTableSettings; + run_events: RetentionRunEventsSettings; +} + +export interface RetentionTableSweepResult { + deleted: number; + backlog: boolean; + error: string; +} + +export interface RetentionSweptTableResult extends RetentionTableSweepResult { + previewed: boolean; + would_delete: number; +} + +export interface RetentionLastSweep { + started_at: string; + finished_at: string; + office_routine_runs: RetentionSweptTableResult; + runs: RetentionSweptTableResult; + run_events: RetentionTableSweepResult; + route_attempts: RetentionTableSweepResult; + run_skills: RetentionTableSweepResult; +} + +export type RetentionCensusState = "not_computed" | "fresh" | "stale"; + +export interface RetentionTableCensus { + state: RetentionCensusState; + retained_count: number; + as_of: string; + unknown_statuses?: string[]; + top_routine_id?: string; + top_routine_share?: number; +} + +export interface RetentionRetainedCounts { + office_routine_runs: RetentionTableCensus; + runs: RetentionTableCensus; + run_events: RetentionTableCensus; +} + +export interface RetentionStatus { + settings: RetentionSettings; + last_sweep: RetentionLastSweep | null; + skip_count: number; + last_skip_at?: string; + retained_counts: RetentionRetainedCounts; +} + export interface RestartCapability { supported: boolean; mode: "manual" | "supervisor" | string; diff --git a/apps/web/src/locales/en/system.json b/apps/web/src/locales/en/system.json index 63bda933697..b103768c767 100644 --- a/apps/web/src/locales/en/system.json +++ b/apps/web/src/locales/en/system.json @@ -207,8 +207,52 @@ "navFeatureToggles": "Feature Toggles", "navLicenses": "Licenses", "navLogs": "Logs", + "navRetention": "Run History Retention", "navUpdates": "Updates", "navUsers": "Users", + "retentionPageDescription": "Bounds how long Office run history (office_routine_runs, runs, and their satellite tables) stays in the database.", + "retentionAdminOnly": "Only an admin can change retention settings.", + "retentionLoadFailed": "Retention settings could not be loaded.", + "retentionSaveFailed": "Retention settings could not be saved.", + "retentionPolicyTitle": "Retention policy", + "retentionPolicyDescription": "A scheduled sweep, separate from the 5-second Office tick, deletes finished run history once it passes its retention window. Rows that represent current work in progress are never deleted, regardless of these settings.", + "retentionEnabledLabel": "Delete eligible run history", + "retentionEnabledDescription": "When off, Kandev never deletes office_routine_runs or runs rows. Retained-row counts below keep updating so you can see backlog build up before turning deletion back on.", + "retentionSweepIntervalLabel": "Sweep interval (hours)", + "retentionSweepIntervalHelp": "How often the sweep runs, from 1 to 168 hours (1 week).", + "retentionBatchLimitLabel": "Rows deleted per sweep", + "retentionBatchLimitHelp": "Maximum rows deleted per table in one sweep, from 100 to 100,000. Lower this if a sweep competes with other database load; a large backlog is cleared over several sweeps instead of one.", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "Routine execution records: one row per completed or failed routine run.", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "Task run records: one row per completed, failed, or cancelled run.", + "retentionRunEventsSectionTitle": "run_events and satellite tables", + "retentionRunEventsSectionDescription": "run_events, office_run_route_attempts, and office_run_skills rows are deleted together with the run that owns them; they have no window or floor of their own.", + "retentionWindowDaysLabel": "Retention window (days)", + "retentionWindowDaysHelp": "Finished rows older than this many days become eligible for deletion, from 1 to 3650 days.", + "retentionFloorPerOwnerLabel": "Minimum kept per owner", + "retentionFloorPerOwnerHelp": "Always keeps at least this many of each owner's most recent finished rows, even past the retention window, from 0 to 10,000. Set to 0 to disable the floor.", + "retentionWarnRowsLabel": "Warn above this many retained rows", + "retentionWarnRowsHelp": "Raises a health warning once retained rows exceed this count. Set to 0 to disable the warning.", + "retentionRunEventsWarnRowsHelp": "Raises a health warning once total retained run_events rows exceed this count. Set to 0 to disable the warning.", + "retentionStatusTitle": "Retention status", + "retentionStatusDescription": "The most recent sweep outcome and the current retained-row counts for each thresholded table.", + "retentionNeverSweptMessage": "No sweep has run yet since the backend last started.", + "retentionSweepStartedAtLabel": "Started", + "retentionSweepFinishedAtLabel": "Finished", + "retentionDeletedLabel": "Deleted", + "retentionWouldDeleteLabel": "Would delete (preview)", + "retentionBacklogLabel": "Backlog: more eligible rows remain past this sweep's batch limit", + "retentionTableErrorLabel": "Error", + "retentionSkipCountLabel": "Sweeps skipped", + "retentionSkipCountHelp": "Counts sweeps that were due but did not run, most often because another Kandev process held the retention lock at the same time.", + "retentionLastSkipAtLabel": "Last skipped", + "retentionRetainedCountsTitle": "Retained rows", + "retentionCensusNotComputed": "Not yet measured", + "retentionCensusStale": "Stale: last measurement failed, showing the last successful count", + "retentionCensusAsOfLabel": "As of", + "retentionUnknownStatusesLabel": "Rows with an unrecognized status (not counted as history or live state)", + "retentionTopRoutineShareLabel": "Largest single routine's share of retained rows", "backendReloadRequiredTitle": "Reload required", "backendReloadRequiredBody": "Kandev restarted. Reload this page to continue. Reloading discards unsaved changes.", "backendReloadRequiredAction": "Reload page", diff --git a/apps/web/src/locales/pseudo/system.json b/apps/web/src/locales/pseudo/system.json index 7711697a8e3..f90f569e914 100644 --- a/apps/web/src/locales/pseudo/system.json +++ b/apps/web/src/locales/pseudo/system.json @@ -207,8 +207,52 @@ "navFeatureToggles": "Ƒēàţũŕē Ţōĝĝĺēś", "navLicenses": "Ĺĩćēńśēś", "navLogs": "Ĺōĝś", + "navRetention": "Ŕũń Ĥĩśţōŕŷ Ŕēţēńţĩōń", "navUpdates": "Ũƥďàţēś", "navUsers": "Ũśēŕś", + "retentionPageDescription": "Ɓōũńďś ĥōŵ ĺōńĝ Ōƒƒĩćē ŕũń ĥĩśţōŕŷ (ōƒƒĩćē_ŕōũţĩńē_ŕũńś, ŕũńś, àńď ţĥēĩŕ śàţēĺĺĩţē ţàƀĺēś) śţàŷś ĩń ţĥē ďàţàƀàśē.", + "retentionAdminOnly": "Ōńĺŷ àń àďḿĩń ćàń ćĥàńĝē ŕēţēńţĩōń śēţţĩńĝś.", + "retentionLoadFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē ĺōàďēď.", + "retentionSaveFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē śàvēď.", + "retentionPolicyTitle": "Ŕēţēńţĩōń ƥōĺĩćŷ", + "retentionPolicyDescription": "À śćĥēďũĺēď śŵēēƥ, śēƥàŕàţē ƒŕōḿ ţĥē 5-śēćōńď Ōƒƒĩćē ţĩćķ, ďēĺēţēś ƒĩńĩśĥēď ŕũń ĥĩśţōŕŷ ōńćē ĩţ ƥàśśēś ĩţś ŕēţēńţĩōń ŵĩńďōŵ. Ŕōŵś ţĥàţ ŕēƥŕēśēńţ ćũŕŕēńţ ŵōŕķ ĩń ƥŕōĝŕēśś àŕē ńēvēŕ ďēĺēţēď, ŕēĝàŕďĺēśś ōƒ ţĥēśē śēţţĩńĝś.", + "retentionEnabledLabel": "Ďēĺēţē ēĺĩĝĩƀĺē ŕũń ĥĩśţōŕŷ", + "retentionEnabledDescription": "Ŵĥēń ōƒƒ, Ķàńďēv ńēvēŕ ďēĺēţēś ōƒƒĩćē_ŕōũţĩńē_ŕũńś ōŕ ŕũńś ŕōŵś. Ŕēţàĩńēď-ŕōŵ ćōũńţś ƀēĺōŵ ķēēƥ ũƥďàţĩńĝ śō ŷōũ ćàń śēē ƀàćķĺōĝ ƀũĩĺď ũƥ ƀēƒōŕē ţũŕńĩńĝ ďēĺēţĩōń ƀàćķ ōń.", + "retentionSweepIntervalLabel": "Śŵēēƥ ĩńţēŕvàĺ (ĥōũŕś)", + "retentionSweepIntervalHelp": "Ĥōŵ ōƒţēń ţĥē śŵēēƥ ŕũńś, ƒŕōḿ 1 ţō 168 ĥōũŕś (1 ŵēēķ).", + "retentionBatchLimitLabel": "Ŕōŵś ďēĺēţēď ƥēŕ śŵēēƥ", + "retentionBatchLimitHelp": "Ḿàxĩḿũḿ ŕōŵś ďēĺēţēď ƥēŕ ţàƀĺē ĩń ōńē śŵēēƥ, ƒŕōḿ 100 ţō 100,000. Ĺōŵēŕ ţĥĩś ĩƒ à śŵēēƥ ćōḿƥēţēś ŵĩţĥ ōţĥēŕ ďàţàƀàśē ĺōàď; à ĺàŕĝē ƀàćķĺōĝ ĩś ćĺēàŕēď ōvēŕ śēvēŕàĺ śŵēēƥś ĩńśţēàď ōƒ ōńē.", + "retentionRoutineRunsSectionTitle": "ōƒƒĩćē_ŕōũţĩńē_ŕũńś", + "retentionRoutineRunsSectionDescription": "Ŕōũţĩńē ēxēćũţĩōń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď ōŕ ƒàĩĺēď ŕōũţĩńē ŕũń.", + "retentionRunsSectionTitle": "ŕũńś", + "retentionRunsSectionDescription": "Ţàśķ ŕũń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď, ƒàĩĺēď, ōŕ ćàńćēĺĺēď ŕũń.", + "retentionRunEventsSectionTitle": "ŕũń_ēvēńţś àńď śàţēĺĺĩţē ţàƀĺēś", + "retentionRunEventsSectionDescription": "ŕũń_ēvēńţś, ōƒƒĩćē_ŕũń_ŕōũţē_àţţēḿƥţś, àńď ōƒƒĩćē_ŕũń_śķĩĺĺś ŕōŵś àŕē ďēĺēţēď ţōĝēţĥēŕ ŵĩţĥ ţĥē ŕũń ţĥàţ ōŵńś ţĥēḿ; ţĥēŷ ĥàvē ńō ŵĩńďōŵ ōŕ ƒĺōōŕ ōƒ ţĥēĩŕ ōŵń.", + "retentionWindowDaysLabel": "Ŕēţēńţĩōń ŵĩńďōŵ (ďàŷś)", + "retentionWindowDaysHelp": "Ƒĩńĩśĥēď ŕōŵś ōĺďēŕ ţĥàń ţĥĩś ḿàńŷ ďàŷś ƀēćōḿē ēĺĩĝĩƀĺē ƒōŕ ďēĺēţĩōń, ƒŕōḿ 1 ţō 3650 ďàŷś.", + "retentionFloorPerOwnerLabel": "Ḿĩńĩḿũḿ ķēƥţ ƥēŕ ōŵńēŕ", + "retentionFloorPerOwnerHelp": "Àĺŵàŷś ķēēƥś àţ ĺēàśţ ţĥĩś ḿàńŷ ōƒ ēàćĥ ōŵńēŕ'ś ḿōśţ ŕēćēńţ ƒĩńĩśĥēď ŕōŵś, ēvēń ƥàśţ ţĥē ŕēţēńţĩōń ŵĩńďōŵ, ƒŕōḿ 0 ţō 10,000. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ƒĺōōŕ.", + "retentionWarnRowsLabel": "Ŵàŕń àƀōvē ţĥĩś ḿàńŷ ŕēţàĩńēď ŕōŵś", + "retentionWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ŕēţàĩńēď ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.", + "retentionRunEventsWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ţōţàĺ ŕēţàĩńēď ŕũń_ēvēńţś ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.", + "retentionStatusTitle": "Ŕēţēńţĩōń śţàţũś", + "retentionStatusDescription": "Ţĥē ḿōśţ ŕēćēńţ śŵēēƥ ōũţćōḿē àńď ţĥē ćũŕŕēńţ ŕēţàĩńēď-ŕōŵ ćōũńţś ƒōŕ ēàćĥ ţĥŕēśĥōĺďēď ţàƀĺē.", + "retentionNeverSweptMessage": "Ńō śŵēēƥ ĥàś ŕũń ŷēţ śĩńćē ţĥē ƀàćķēńď ĺàśţ śţàŕţēď.", + "retentionSweepStartedAtLabel": "Śţàŕţēď", + "retentionSweepFinishedAtLabel": "Ƒĩńĩśĥēď", + "retentionDeletedLabel": "Ďēĺēţēď", + "retentionWouldDeleteLabel": "Ŵōũĺď ďēĺēţē (ƥŕēvĩēŵ)", + "retentionBacklogLabel": "Ɓàćķĺōĝ: ḿōŕē ēĺĩĝĩƀĺē ŕōŵś ŕēḿàĩń ƥàśţ ţĥĩś śŵēēƥ'ś ƀàţćĥ ĺĩḿĩţ", + "retentionTableErrorLabel": "Ēŕŕōŕ", + "retentionSkipCountLabel": "Śŵēēƥś śķĩƥƥēď", + "retentionSkipCountHelp": "Ćōũńţś śŵēēƥś ţĥàţ ŵēŕē ďũē ƀũţ ďĩď ńōţ ŕũń, ḿōśţ ōƒţēń ƀēćàũśē àńōţĥēŕ Ķàńďēv ƥŕōćēśś ĥēĺď ţĥē ŕēţēńţĩōń ĺōćķ àţ ţĥē śàḿē ţĩḿē.", + "retentionLastSkipAtLabel": "Ĺàśţ śķĩƥƥēď", + "retentionRetainedCountsTitle": "Ŕēţàĩńēď ŕōŵś", + "retentionCensusNotComputed": "Ńōţ ŷēţ ḿēàśũŕēď", + "retentionCensusStale": "Śţàĺē: ĺàśţ ḿēàśũŕēḿēńţ ƒàĩĺēď, śĥōŵĩńĝ ţĥē ĺàśţ śũććēśśƒũĺ ćōũńţ", + "retentionCensusAsOfLabel": "Àś ōƒ", + "retentionUnknownStatusesLabel": "Ŕōŵś ŵĩţĥ àń ũńŕēćōĝńĩźēď śţàţũś (ńōţ ćōũńţēď àś ĥĩśţōŕŷ ōŕ ĺĩvē śţàţē)", + "retentionTopRoutineShareLabel": "Ĺàŕĝēśţ śĩńĝĺē ŕōũţĩńē'ś śĥàŕē ōƒ ŕēţàĩńēď ŕōŵś", "backendReloadRequiredTitle": "Ŕēĺōàď ŕēqũĩŕēď", "backendReloadRequiredBody": "Ķàńďēv ŕēśţàŕţēď. Ŕēĺōàď ţĥĩś ƥàĝē ţō ćōńţĩńũē. Ŕēĺōàďĩńĝ ďĩśćàŕďś ũńśàvēď ćĥàńĝēś.", "backendReloadRequiredAction": "Ŕēĺōàď ƥàĝē", diff --git a/apps/web/src/locales/pt-pt/system.json b/apps/web/src/locales/pt-pt/system.json index dc9122adfaf..097df6de1ca 100644 --- a/apps/web/src/locales/pt-pt/system.json +++ b/apps/web/src/locales/pt-pt/system.json @@ -202,8 +202,52 @@ "navFeatureToggles": "Interruptores de funcionalidades", "navLicenses": "Licenças", "navLogs": "Registos", + "navRetention": "Retenção do histórico de execuções", "navUpdates": "Atualizações", "navUsers": "Utilizadores", + "retentionPageDescription": "Limita durante quanto tempo o histórico de execuções do Office (office_routine_runs, runs e as respetivas tabelas satélite) permanece na base de dados.", + "retentionAdminOnly": "Só um administrador pode alterar as definições de retenção.", + "retentionLoadFailed": "Não foi possível carregar as definições de retenção.", + "retentionSaveFailed": "Não foi possível guardar as definições de retenção.", + "retentionPolicyTitle": "Política de retenção", + "retentionPolicyDescription": "Uma limpeza agendada, separada do ciclo do Office de 5 segundos, elimina o histórico de execuções concluído assim que ultrapassa a sua janela de retenção. As linhas que representam trabalho em curso nunca são eliminadas, independentemente destas definições.", + "retentionEnabledLabel": "Eliminar histórico de execuções elegível", + "retentionEnabledDescription": "Quando desativado, o Kandev nunca elimina linhas de office_routine_runs ou runs. As contagens de linhas retidas abaixo continuam a ser atualizadas para poder ver a acumulação de atrasos antes de reativar a eliminação.", + "retentionSweepIntervalLabel": "Intervalo de limpeza (horas)", + "retentionSweepIntervalHelp": "Frequência com que a limpeza é executada, entre 1 e 168 horas (1 semana).", + "retentionBatchLimitLabel": "Linhas eliminadas por limpeza", + "retentionBatchLimitHelp": "Número máximo de linhas eliminadas por tabela numa limpeza, entre 100 e 100 000. Reduza este valor se a limpeza competir com outra carga na base de dados; um atraso grande é eliminado ao longo de várias limpezas em vez de uma só.", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "Registos de execução de rotinas: uma linha por cada execução de rotina concluída ou falhada.", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "Registos de execução de tarefas: uma linha por cada execução concluída, falhada ou cancelada.", + "retentionRunEventsSectionTitle": "run_events e tabelas satélite", + "retentionRunEventsSectionDescription": "As linhas de run_events, office_run_route_attempts e office_run_skills são eliminadas em conjunto com a execução que as possui; não têm janela nem limite mínimo próprios.", + "retentionWindowDaysLabel": "Janela de retenção (dias)", + "retentionWindowDaysHelp": "As linhas concluídas com mais deste número de dias tornam-se elegíveis para eliminação, entre 1 e 3650 dias.", + "retentionFloorPerOwnerLabel": "Mínimo mantido por proprietário", + "retentionFloorPerOwnerHelp": "Mantém sempre pelo menos este número de linhas concluídas mais recentes de cada proprietário, mesmo além da janela de retenção, entre 0 e 10 000. Defina 0 para desativar este limite mínimo.", + "retentionWarnRowsLabel": "Avisar acima deste número de linhas retidas", + "retentionWarnRowsHelp": "Gera um aviso de saúde quando as linhas retidas ultrapassam este número. Defina 0 para desativar o aviso.", + "retentionRunEventsWarnRowsHelp": "Gera um aviso de saúde quando o total de linhas de run_events retidas ultrapassa este número. Defina 0 para desativar o aviso.", + "retentionStatusTitle": "Estado da retenção", + "retentionStatusDescription": "O resultado da limpeza mais recente e as contagens atuais de linhas retidas para cada tabela com limiar definido.", + "retentionNeverSweptMessage": "Ainda não foi executada nenhuma limpeza desde o último arranque do backend.", + "retentionSweepStartedAtLabel": "Iniciada", + "retentionSweepFinishedAtLabel": "Concluída", + "retentionDeletedLabel": "Eliminadas", + "retentionWouldDeleteLabel": "Seriam eliminadas (pré-visualização)", + "retentionBacklogLabel": "Atraso: ainda há mais linhas elegíveis além do limite desta limpeza", + "retentionTableErrorLabel": "Erro", + "retentionSkipCountLabel": "Limpezas ignoradas", + "retentionSkipCountHelp": "Conta as limpezas que deviam ter sido executadas mas não o foram, geralmente porque outro processo do Kandev detinha o bloqueio de retenção ao mesmo tempo.", + "retentionLastSkipAtLabel": "Última ignorada em", + "retentionRetainedCountsTitle": "Linhas retidas", + "retentionCensusNotComputed": "Ainda não medido", + "retentionCensusStale": "Desatualizado: a última medição falhou, a mostrar a última contagem bem-sucedida", + "retentionCensusAsOfLabel": "Referente a", + "retentionUnknownStatusesLabel": "Linhas com um estado não reconhecido (não contadas como histórico nem como estado ativo)", + "retentionTopRoutineShareLabel": "Proporção de linhas retidas pertencentes à maior rotina individual", "backendReloadRequiredTitle": "É necessário recarregar", "backendReloadRequiredBody": "O Kandev foi reiniciado. Recarregue esta página para continuar. O recarregamento elimina as alterações não guardadas.", "backendReloadRequiredAction": "Recarregar página", diff --git a/apps/web/src/locales/zh-cn/system.json b/apps/web/src/locales/zh-cn/system.json index 333cbe57e8c..1361e839d65 100644 --- a/apps/web/src/locales/zh-cn/system.json +++ b/apps/web/src/locales/zh-cn/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能开关", "navLicenses": "许可证", "navLogs": "日志", + "navRetention": "运行历史保留", "navUpdates": "更新", "navUsers": "用户", + "retentionPageDescription": "限制 Office 运行历史(office_routine_runs、runs 及其附属表)在数据库中保留的时长。", + "retentionAdminOnly": "只有管理员才能更改保留设置。", + "retentionLoadFailed": "无法加载保留设置。", + "retentionSaveFailed": "无法保存保留设置。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一个与 5 秒 Office 心跳独立的定时清理任务,会在已完成的运行历史超出其保留窗口后将其删除。代表当前进行中工作的行永远不会被删除,与这些设置无关。", + "retentionEnabledLabel": "删除符合条件的运行历史", + "retentionEnabledDescription": "关闭时,Kandev 永远不会删除 office_routine_runs 或 runs 中的行。下方的保留行数仍会持续更新,方便你在重新启用删除前查看积压情况。", + "retentionSweepIntervalLabel": "清理间隔(小时)", + "retentionSweepIntervalHelp": "清理任务运行的频率,介于 1 到 168 小时(1 周)之间。", + "retentionBatchLimitLabel": "每次清理删除的行数", + "retentionBatchLimitHelp": "每次清理中每张表最多删除的行数,介于 100 到 100000 之间。如果清理与其他数据库负载相互争用,可以调低此值;较大的积压会分多次清理逐步处理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任务执行记录:每完成或失败一次例行任务运行对应一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任务运行记录:每完成、失败或取消一次运行对应一行。", + "retentionRunEventsSectionTitle": "run_events 及附属表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行会随其所属的运行一起删除;它们没有自己的窗口或下限设置。", + "retentionWindowDaysLabel": "保留窗口(天)", + "retentionWindowDaysHelp": "已完成且超过此天数的行将符合删除条件,介于 1 到 3650 天之间。", + "retentionFloorPerOwnerLabel": "每个所有者的最少保留数", + "retentionFloorPerOwnerHelp": "始终至少保留每个所有者最近完成的这么多行,即使超出保留窗口,介于 0 到 10000 之间。设为 0 可停用此下限。", + "retentionWarnRowsLabel": "超过此保留行数时发出警告", + "retentionWarnRowsHelp": "当保留行数超过此数值时触发健康警告。设为 0 可停用该警告。", + "retentionRunEventsWarnRowsHelp": "当保留的 run_events 总行数超过此数值时触发健康警告。设为 0 可停用该警告。", + "retentionStatusTitle": "保留状态", + "retentionStatusDescription": "最近一次清理的结果,以及每张设有阈值的表当前的保留行数。", + "retentionNeverSweptMessage": "自后端上次启动以来,尚未运行过清理任务。", + "retentionSweepStartedAtLabel": "开始于", + "retentionSweepFinishedAtLabel": "结束于", + "retentionDeletedLabel": "已删除", + "retentionWouldDeleteLabel": "预计删除(预览)", + "retentionBacklogLabel": "积压:超出本次清理批量上限的符合条件的行仍然存在", + "retentionTableErrorLabel": "错误", + "retentionSkipCountLabel": "已跳过的清理次数", + "retentionSkipCountHelp": "统计原本应该运行但未运行的清理次数,通常是因为同一时间有另一个 Kandev 进程持有保留锁。", + "retentionLastSkipAtLabel": "上次跳过于", + "retentionRetainedCountsTitle": "保留的行数", + "retentionCensusNotComputed": "尚未统计", + "retentionCensusStale": "已过期:上次统计失败,显示的是上一次成功的计数", + "retentionCensusAsOfLabel": "统计时间", + "retentionUnknownStatusesLabel": "状态无法识别的行(既不计入历史,也不计入当前状态)", + "retentionTopRoutineShareLabel": "占保留行数比例最高的单个例行任务", "backendReloadRequiredTitle": "需要重新加载", "backendReloadRequiredBody": "Kandev 已重启。请重新加载此页面以继续。重新加载会丢弃未保存的更改。", "backendReloadRequiredAction": "重新加载页面", diff --git a/apps/web/src/locales/zh-hk/system.json b/apps/web/src/locales/zh-hk/system.json index 38563613bbe..3f65e70e084 100644 --- a/apps/web/src/locales/zh-hk/system.json +++ b/apps/web/src/locales/zh-hk/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能開關", "navLicenses": "許可證", "navLogs": "日誌", + "navRetention": "執行歷史保留", "navUpdates": "更新", "navUsers": "用戶", + "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在數據庫中保留的時長。", + "retentionAdminOnly": "只有管理員才能更改保留設定。", + "retentionLoadFailed": "無法載入保留設定。", + "retentionSaveFailed": "無法儲存保留設定。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留窗口後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。", + "retentionEnabledLabel": "刪除符合條件的執行歷史", + "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啓用刪除前查看積壓情況。", + "retentionSweepIntervalLabel": "清理間隔(小時)", + "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。", + "retentionBatchLimitLabel": "每次清理刪除的行數", + "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他數據庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。", + "retentionRunEventsSectionTitle": "run_events 及附屬表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的窗口或下限設定。", + "retentionWindowDaysLabel": "保留窗口(天)", + "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。", + "retentionFloorPerOwnerLabel": "每個所有者的最少保留數", + "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留窗口,介於 0 到 10000 之間。設為 0 可停用此下限。", + "retentionWarnRowsLabel": "超過此保留行數時發出警告", + "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionStatusTitle": "保留狀態", + "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。", + "retentionNeverSweptMessage": "自後端上次啓動以來,尚未執行過清理任務。", + "retentionSweepStartedAtLabel": "開始於", + "retentionSweepFinishedAtLabel": "結束於", + "retentionDeletedLabel": "已刪除", + "retentionWouldDeleteLabel": "預計刪除(預覽)", + "retentionBacklogLabel": "積壓:超出本次清理批量上限的符合條件的行仍然存在", + "retentionTableErrorLabel": "錯誤", + "retentionSkipCountLabel": "已跳過的清理次數", + "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 程序持有保留鎖。", + "retentionLastSkipAtLabel": "上次跳過於", + "retentionRetainedCountsTitle": "保留的行數", + "retentionCensusNotComputed": "尚未統計", + "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數", + "retentionCensusAsOfLabel": "統計時間", + "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)", + "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務", "backendReloadRequiredTitle": "需要重新載入", "backendReloadRequiredBody": "Kandev 已重啓。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。", "backendReloadRequiredAction": "重新載入頁面", diff --git a/apps/web/src/locales/zh-tw/system.json b/apps/web/src/locales/zh-tw/system.json index 8aec9747198..192eec0cddb 100644 --- a/apps/web/src/locales/zh-tw/system.json +++ b/apps/web/src/locales/zh-tw/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能開關", "navLicenses": "許可證", "navLogs": "日誌", + "navRetention": "執行歷史保留", "navUpdates": "更新", "navUsers": "使用者", + "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在資料庫中保留的時長。", + "retentionAdminOnly": "只有管理員才能更改保留設定。", + "retentionLoadFailed": "無法載入保留設定。", + "retentionSaveFailed": "無法儲存保留設定。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留視窗後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。", + "retentionEnabledLabel": "刪除符合條件的執行歷史", + "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啟用刪除前檢視積壓情況。", + "retentionSweepIntervalLabel": "清理間隔(小時)", + "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。", + "retentionBatchLimitLabel": "每次清理刪除的行數", + "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他資料庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。", + "retentionRunEventsSectionTitle": "run_events 及附屬表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的視窗或下限設定。", + "retentionWindowDaysLabel": "保留視窗(天)", + "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。", + "retentionFloorPerOwnerLabel": "每個所有者的最少保留數", + "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留視窗,介於 0 到 10000 之間。設為 0 可停用此下限。", + "retentionWarnRowsLabel": "超過此保留行數時發出警告", + "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionStatusTitle": "保留狀態", + "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。", + "retentionNeverSweptMessage": "自後端上次啟動以來,尚未執行過清理任務。", + "retentionSweepStartedAtLabel": "開始於", + "retentionSweepFinishedAtLabel": "結束於", + "retentionDeletedLabel": "已刪除", + "retentionWouldDeleteLabel": "預計刪除(預覽)", + "retentionBacklogLabel": "積壓:超出本次清理批次上限的符合條件的行仍然存在", + "retentionTableErrorLabel": "錯誤", + "retentionSkipCountLabel": "已跳過的清理次數", + "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 處理程序持有保留鎖。", + "retentionLastSkipAtLabel": "上次跳過於", + "retentionRetainedCountsTitle": "保留的行數", + "retentionCensusNotComputed": "尚未統計", + "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數", + "retentionCensusAsOfLabel": "統計時間", + "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)", + "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務", "backendReloadRequiredTitle": "需要重新載入", "backendReloadRequiredBody": "Kandev 已重啟。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。", "backendReloadRequiredAction": "重新載入頁面", diff --git a/docs/specs/office/requirements/run-history-retention-operations.md b/docs/specs/office/requirements/run-history-retention-operations.md new file mode 100644 index 00000000000..394f70290e9 --- /dev/null +++ b/docs/specs/office/requirements/run-history-retention-operations.md @@ -0,0 +1,214 @@ +--- +status: draft +system: office +created: 2026-09-09 +owners: + - kandev +--- + +# Office Run History Retention Operations Requirements + +## Overview + +[Run history retention](run-history-retention.md) defines which Office run +history rows may be deleted and how the sweep that deletes them behaves. This +document defines the other half of that contract: what an operator is told +before the first row is removed, what they are told while the tables grow, what +they can configure, and what they can read about the last sweep. + +The split follows the ownership boundary the retention contract already draws. +Office owns the rows and the deletion policy because they are Office primitives. +The System pages own the operator surface those decisions are reported through, +and that surface is a separate contract with a separate consumer: an operator +reading a settings page and a health card, rather than a scheduler deleting +rows. Splitting here keeps each document inside its size limit without cutting +either contract. + +The requirement IDs continue the retention capability's sequence rather than +starting a new one, because these are the same capability's requirements viewed +from the operator's side. + +## Terminology + +Terms are defined once, in +[run history retention](run-history-retention.md#terminology), and used here +with the same meaning. The ones this document leans on most: + +- **Retention sweep**, **preview**, and **retained count**. +- **Swept tables** (`office_routine_runs`, `runs`), **reported tables** (those + two plus the three run satellites), and **thresholded tables** + (`office_routine_runs`, `runs`, `run_events`). "Per table" always names one of + these three sets, never an unqualified "table". + +## Requirements + +### REQ-OFFICE-RUN-HISTORY-RETENTION-003: Warn before deleting, and warn while growing + +**Intent:** An operator learns what retention is about to remove before it +removes anything, and learns that history is growing past a threshold while +there is still time to widen the window or fix the routine. + +As an operator upgrading an install with a year of run history, I want to be +told what the first sweep would delete before it deletes it, so that I can widen +the retention window first if that history matters to me. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.1:** The first evaluation of each swept table + on a database shall be a preview for that table: it evaluates the policy, + reports the number of rows it would delete from that table, and deletes + nothing from it. A table that has not completed a preview never deletes, + regardless of what any other table has done. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.2:** When a preview sweep would delete + at least one row, the system shall emit an operator-visible warning naming + each table, its would-delete count, and the configured retention window, and + stating that deletion begins at the next scheduled sweep. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.3:** When a preview sweep would delete + no rows, the system shall not emit a warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.4:** The system shall run a preview at most + once per swept table per database. A table's preview shall be recorded only + when that table's preview evaluation completed successfully; a table that + failed during a sweep in which it was being previewed shall be previewed again + on the next sweep rather than deleting. Disabling and re-enabling retention, + restarting the backend, or changing any retention setting shall not produce a + second preview for a table that has already completed one. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.5:** When a thresholded table's retained + count exceeds that table's configured warning threshold, the system shall emit + an operator-visible warning naming the table, its retained count, and the + threshold. For `office_routine_runs`, the warning shall also name the routine + holding the largest share of retained rows and that share, where the share is + that routine's retained rows as a proportion of the table's retained count. + When two routines hold an equal largest share, the lower routine identifier + shall be named. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.6:** When a table reports remaining + backlog after a sweep, the system shall emit an operator-visible warning + naming the table, stating that retention is behind, and reporting the number + of rows deleted in that sweep. A table that was previewed in that sweep shall + not produce this warning however many rows were eligible, because a preview + deletes nothing by design and retention is therefore not behind; the preview + warning of AC-OFFICE-RUN-HISTORY-RETENTION-003.2 reports its eligible count + instead. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.7:** When retention is disabled and a + table's retained count exceeds its warning threshold, the system shall emit + the same threshold warning and shall additionally state that retention is + disabled, so a silent unbounded table is distinguishable from one being + actively managed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.8:** Every warning in this requirement + shall be emitted through a channel available in a production build, and shall + not be observable only through the debug metrics endpoint. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.9:** A preview sweep's would-delete + counts shall be the full eligible count per table, not capped by the batch + limit, so an operator is told the real size of what is about to be removed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.10:** When a swept table's recorded preview + state cannot be read or parsed, the system shall treat that table as not yet + previewed and shall emit an operator-visible warning naming the table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.11:** Retained counts for the thresholded + tables shall be produced by a count evaluation that does not require a sweep + to have run, so the counts and the threshold warnings required by AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and + AC-OFFICE-RUN-HISTORY-RETENTION-004.8 are available on a fresh install and while retention is disabled. The + counts shall be refreshed on the sweep interval and reused between refreshes + rather than recomputed for each operator page load or health poll. The first + count evaluation shall run at startup, before the delay of + AC-OFFICE-RUN-HISTORY-RETENTION-002.10 arms the first sweep, so an operator is + not shown countless tables for a sweep interval after every restart. When a + count evaluation fails, the system shall keep the counts from the last + successful evaluation, report them as stale together with the time they were + produced, and emit an operator-visible warning. A failed count shall not + suppress the threshold warnings derived from the last successful counts, and + shall not fail the sweep. Until the first evaluation has completed, and when it + fails with no earlier successful evaluation to fall back on, the system shall + report the counts as not yet computed and shall not render them as zero, on the + same grounds as AC-OFFICE-RUN-HISTORY-RETENTION-004.7: a count nobody has taken + must not read as a table that is empty. Success and failure shall be tracked + per thresholded table, so one table's failed count neither discards nor marks + stale another table's successful one. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-004: Operator configuration and sweep visibility + +**Intent:** Retention is configurable, its values are validated, and the result +of the most recent sweep is readable by an operator without reading logs. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.1:** An operator can read and change + whether retention is enabled, the retention window per in-scope table, the + sweep interval, the retention floor per in-scope table, the batch limit, and + the warning threshold per in-scope table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.2:** Retention shall be enabled by + default. Default values shall be: retention window 30 days for both + `office_routine_runs` and `runs`; sweep interval 6 hours; retention floor 50 + rows per owner for both tables; batch limit 5,000 rows per table per sweep; + warning threshold 25,000 rows for `office_routine_runs`, 25,000 for `runs`, + and 250,000 for `run_events`. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.3:** The system shall reject a setting + outside its permitted range with an error naming the field, and shall leave + the stored settings unchanged. Permitted ranges are: retention window 1 to + 3,650 days; sweep interval 1 to 168 hours; retention floor 0 to 10,000; batch + limit 100 to 100,000; warning threshold 0 or greater, where 0 disables that + table's threshold warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.4:** When stored retention settings + cannot be read or parsed, the system shall use the documented defaults, emit + an operator-visible warning, and continue. Unreadable settings shall not + disable retention silently and shall not fail startup. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.5:** A settings change shall take effect + without a backend restart, and shall apply from the next scheduled sweep + rather than interrupting a sweep in progress. Two concurrent writes resolve + last-writer-wins; repeating an identical write changes nothing and returns the + same normalized document. A sweep shall read the stored settings at its start + rather than relying on a value cached when this process last observed a change, + so that where several backend processes share one database a process that did + not serve the write still sweeps under the new settings rather than the + replaced ones. When that read fails or the stored settings cannot be parsed, + the sweep shall be skipped and recorded as skipped rather than run against the + documented defaults, because a default window is shorter than a window an + operator has widened and sweeping under it would delete the history they + configured the system to keep. This does not change + AC-OFFICE-RUN-HISTORY-RETENTION-004.4, which governs reading settings for + reporting and startup, where using the defaults deletes nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.6:** An operator can read, on the System + pages, the outcome of the most recent completed sweep: when it ran, which + swept tables it previewed, the rows deleted per reported table, the rows it + would have deleted per previewed table, the retained count per thresholded + table, whether any table has remaining backlog, and any table that failed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.7:** When no sweep has run since the + backend started, the surface in AC-OFFICE-RUN-HISTORY-RETENTION-004.6 shall + say so explicitly rather than render an empty or zeroed result that reads as a + sweep that deleted nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.8:** The sweep-result surface shall be + readable while retention is disabled, reporting retained counts and the + disabled state. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.9:** A settings write shall replace the whole + settings document. A field the caller omits shall take its documented default + rather than its previously stored value, so the same request always produces + the same stored document. An unrecognized field, a numeric field whose + value is not a whole number, a field present with a JSON `null`, and a field + whose value is not of its documented type, shall each be rejected with an error + naming the field, leaving the stored settings unchanged. `null` shall be + rejected rather than treated as omission: omission means the documented + default, so reading `null` the same way would let a client silently replace a + configured retention window with a shorter default and destroy history the + operator meant to keep. + +## Out of scope + +- **A persisted history of retention sweeps.** The last sweep's result is held + in memory and reported alongside the durable warnings. A growing table + recording the work of the job that stops tables growing is the same defect in + a new place. +- **Per-workspace or per-routine retention overrides.** Settings are + instance-wide. The retention floor is already per owner, which covers what a + per-routine override would most often be used for. +- **Archival or export before deletion.** Deleted history is gone. An install + needing it kept sets a longer window or takes a backup. +- **Anything the deletion policy owns.** Eligibility, ordering, batching, + atomicity, and engine parity are specified in + [run history retention](run-history-retention.md) and are not restated here. + +## Prior art + +The receipts for both prior-art legs are recorded once, in +[run history retention](run-history-retention.md#prior-art). The finding that +bears on this document specifically: neither surveyed product previews before a +policy's first deletion, nor warns ahead of the window. Those two behaviors are +where this capability goes further, and they are the reason this document +exists rather than being a settings page bolted onto a sweep. diff --git a/docs/specs/office/requirements/run-history-retention.md b/docs/specs/office/requirements/run-history-retention.md new file mode 100644 index 00000000000..2956b949518 --- /dev/null +++ b/docs/specs/office/requirements/run-history-retention.md @@ -0,0 +1,316 @@ +--- +status: draft +system: office +created: 2026-09-09 +owners: + - kandev +--- + +# Office Run History Retention Requirements + +## Overview + +Office writes two families of history rows that nothing removes on a schedule. +`office_routine_runs` records one row per routine firing. `run_events` records +the timeline of every Office run, append-only, alongside the `runs` queue row +and its per-run satellites. The only deletions today are manual and structural: +deleting a routine removes its runs, and deleting a workspace removes everything +belonging to it. An install that never deletes a routine or a workspace grows +forever. + +The growth is driven by a clock, not by usage. A routine on a `*/5 * * * *` +schedule fires 105,120 times a year, and each firing writes a routine-run row +whether or not it does anything. On the reference install, one routine had +accumulated 323 consecutive `coalesced` rows over 28 days while producing no +work at all: that is the curve with the loop broken, and a working loop also +writes the run row, its `run_events` timeline, and its route-attempt rows. + +This document bounds those tables. It defines what is history and may be +deleted, what is live state and must never be deleted on age, when the deletion +runs, and that it behaves identically on both database engines. What an operator +is told before the first row is removed, what they are warned about while the +tables grow, and what they can configure and read is the paired contract in +[run history retention operations](run-history-retention-operations.md). + +Office owns this contract because the rows are defined by Office primitives: the +routine dispatch ledger, the run queue, and the run event timeline. The System +pages own the operator surface it reports through. The filesystem and container +cleanup owned by [storage +maintenance](../../system-page/requirements/storage-maintenance.md) is a +separate capability that never touches database rows. + +## Terminology + +- **Retention sweep** (or **sweep**): one pass that evaluates every in-scope + table against the configured policy and deletes the eligible rows. +- **History row**: a row recording something that already happened, which no + live decision reads. History rows are eligible for deletion. +- **Live-state row**: a row a live decision still reads, regardless of age. + Never eligible for age-based deletion. +- **Run satellite row**: a row keyed by a `runs` row's identifier and owned by + it: a `run_events`, `office_run_route_attempts`, or `office_run_skills` entry. +- **Retention window**: the age past which a history row becomes eligible, + measured from the row's own completion time. +- **Completion time**: `COALESCE(completed_at, created_at)` for a routine-run + row and `COALESCE(finished_at, requested_at)` for a `runs` row. Both fallback + columns are `NOT NULL`, so completion time is never null. +- **Retention floor**: most-recent history rows kept per owner regardless of + age, so a rarely-firing routine or rarely-woken agent keeps visible history. +- **Batch limit**: the maximum rows one sweep deletes from one table. +- **Preview**: a swept table's first evaluation on a database; it counts what it + would delete from that table and deletes nothing. Tracked per swept table. +- **Retained count**: the number of rows a table currently holds — a property of + the table, not of a sweep, defined whether or not a sweep has ever run and + whether or not retention is enabled. +- **Swept tables**: the two tables retention selects rows from by policy, + `office_routine_runs` and `runs`. +- **Reported tables**: the five tables a sweep can delete rows from and reports + deleted counts for: the two swept tables plus `run_events`, + `office_run_route_attempts`, and `office_run_skills`. +- **Thresholded tables**: the three tables carrying a warning threshold, + `office_routine_runs`, `runs`, and `run_events`. + +## Requirements + +### REQ-OFFICE-RUN-HISTORY-RETENTION-001: History is bounded and live state is not + +**Intent:** Bound `office_routine_runs`, `runs`, and the run satellite tables by +age and by an owner-scoped floor, while guaranteeing that no row another +decision still reads is removed because it is old. + +As an operator running Office continuously, I want old run history removed +automatically, so a scheduled routine does not grow the database without limit. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.1:** A routine-run row is a history row + only when its status is one of `skipped`, `coalesced`, `failed`, `done`, or + `cancelled`. When a routine-run row's status is `received` or `task_created`, + the system shall treat it as a live-state row and shall not delete it on age, + at any age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.2:** A `runs` row is a history row when its + status is `finished`, `failed`, or `cancelled`. `cancelled` is terminal: its + only writer moves a row there from `queued` or `claimed` and stamps the + completion timestamp in the same statement. When a `runs` row's status is + `queued` or `claimed`, the system shall treat it as a live-state row and shall + not delete it on age, at any age, including a run parked for a future routing + retry. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.3:** When a history row's completion time is + older than that table's retention window, the system shall make it eligible + for deletion. Completion time is defined in Terminology. A row's + classification as history shall depend on its status alone and shall not + additionally require `completed_at` or `finished_at` to be set; a terminal row + with an unset stamp is history, dated by its fallback column, and ages out + rather than being retained forever. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.4:** The system shall retain the newest + history rows of each owner up to that table's retention floor even when they + are older than the retention window. The owner is the routine for + `office_routine_runs` and the agent profile for `runs`. Newest is completion + time descending, with the row identifier descending as the tiebreak. + Completion time is never null, so this ordering is total on both engines and + does not depend on either engine's default placement of nulls. The identifier + gives a stable order for equal timestamps and is not claimed to be + chronological. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.5:** When the system deletes a `runs` + row, it shall delete that run's satellite rows in the same database + transaction, and no satellite row shall remain that references a `runs` row + the system has deleted. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.6:** The system shall not delete + `run_events` rows for a run it is not deleting in the same transaction, at any + age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.7:** A routine's status shall not exempt + its history from retention. When a routine is `paused`, its history rows are + evaluated by the same policy as an `active` routine's. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.8:** The system shall not delete, alter, + archive, or cancel a task, a session, a task checkout, or a workspace as part + of a retention sweep. A routine-run row naming a task in `linked_task_id` may + be deleted while that task continues to exist. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.9:** A retained `coalesced` routine-run + row may name a `coalesced_into_run_id` whose row has already been deleted. + `coalesced_into_run_id` records provenance and is not a referential + constraint; the system shall not delete a coalesced row because its target was + deleted, and shall not retain a target because a coalesced row names it. + +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.10:** When a row in a swept table holds a + status belonging to neither that table's history set nor its live-state set, + the system shall treat it as a live-state row, shall not delete it at any age, + and shall emit an operator-visible warning naming the table and the + unrecognized status, through the channel required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 in [run history + retention operations](run-history-retention-operations.md). The warning shall be + produced by the count evaluation required by + AC-OFFICE-RUN-HISTORY-RETENTION-003.11 rather than by the deletion path, which + cannot observe a status it does not select; it shall therefore be emitted on an + install where retention is disabled and no sweep runs. When one table holds more + than one unrecognized status, the system shall emit a single warning for that + table naming every unrecognized status in ascending lexicographic order with its + row count. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-002: The retention sweep + +**Intent:** Run retention on its own schedule, off the run-claiming hot path, in +bounded batches, with one sweep at a time and no dependence on database cascade +behavior. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.1:** The system shall run retention on a + dedicated schedule whose interval is configurable in hours, and shall not + perform retention work on the Office run-processing tick. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.2:** When a scheduled sweep is due while a + previous sweep is still running, the system shall skip the due sweep rather + than run two concurrently or queue it, and shall record that it was skipped. + Recording a skip shall not replace the reported result of the last completed + sweep. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.3:** A sweep shall delete at most the batch + limit of rows from `office_routine_runs` and at most that many from `runs`. A + preview evaluation shall never be reported as having remaining backlog: it + deletes nothing by design, so "retention is behind" is not true of it however + many rows were eligible. + The limit counts those rows only; every satellite row of a deleted run is + removed regardless of the limit. When more rows were eligible than the limit + allowed, the system shall complete the sweep, report that table as having + remaining backlog, and continue on the next scheduled sweep. Within a table + the sweep shall select its batch in completion-time ascending order, with the + row identifier ascending as the tiebreak, so the oldest eligible rows are + removed first and both engines select the same batch. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.4:** The system shall re-assert every + eligibility condition in the deletion itself, not only when selecting + candidates. A run that returns to `queued` between selection and deletion, + which a scheduled retry does by clearing `finished_at`, shall not be deleted + by that sweep. The re-asserted conditions shall include the retention floor as + well as status and age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.5:** Each batch shall be atomic: after a + sweep is interrupted by shutdown or error, every batch that was applied is + complete, including its satellite rows, and no batch is partially applied. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.6:** Once a table holds no eligible row, + a further sweep against unchanged settings shall delete nothing from that table + and report zero deletions for it. This criterion is scoped to that drained + state and does not contradict AC-OFFICE-RUN-HISTORY-RETENTION-002.3: while a + table still reports remaining backlog, the next sweep is required to delete its + next batch, so "deletes nothing further" is not claimed of a backlogged table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.7:** When a table's sweep fails with a + database error, the system shall record the failure for that table, continue + with the remaining in-scope tables, and retry the failed table on the next + scheduled sweep. A retention failure shall not fail backend startup, stop the + Office scheduler, or abort the remainder of the sweep. A batch abandoned + because its deletion could not be applied consistently shall be recorded as a + failure for that table rather than as backlog. Every table whose rows that + batch's transaction addressed shall report zero rows deleted for that sweep, so + no table reports rows that the rollback restored. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.8:** When retention is disabled, the + system shall run no sweep and delete no row, and shall still report retained + counts and threshold warnings. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.9:** When every in-scope table is empty + or holds no eligible row, the sweep shall complete reporting zero deletions, + without error and without a warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.10:** The first sweep after the backend + starts shall run after a short fixed delay rather than after a full sweep + interval, so that an install restarted more often than the interval still runs + retention. Later sweeps use the configured interval. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.11:** A sweep shall compute one cutoff + instant at its start and evaluate every table against that instant, so two + tables in one sweep cannot disagree about what "older than the window" means. + +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.12:** On PostgreSQL, where several backend + processes can share one database, sweep exclusivity shall hold across + processes and not only within one: a backend that cannot acquire the retention + lock shall skip its due sweep exactly as it would for a sweep already running + in its own process. On SQLite one backend process owns the database file, so + the in-process guard is sufficient. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.13:** Sweeps shall be scheduled fixed-delay: + the next sweep is armed when the previous one finishes, so a sweep running + longer than the interval delays its successor rather than causing an immediate + second one. A settings change re-arms the delay from the moment of the change. + Enabling retention that was disabled arms the next sweep at the same short + delay as AC-OFFICE-RUN-HISTORY-RETENTION-002.10 rather than at a full interval. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-005: Integrity and database engine parity + +**Intent:** Retention leaves the database consistent and behaves identically on +both supported engines, including where the schema differs between them. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.1:** The system shall produce the same + observable retention outcome on SQLite and on PostgreSQL for the same settings + and the same starting rows: the same rows deleted, the same rows retained, the + same reported counts, and the same warnings. This shall hold on the backlog + path as well, because batch selection order is fixed by named columns in + AC-OFFICE-RUN-HISTORY-RETENTION-002.3 rather than left to the engine. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.2:** The system shall delete satellite + rows explicitly and shall not depend on a foreign-key cascade to remove them. + Neither engine declares a foreign key from a satellite table to `runs`. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.3:** After any sequence of sweeps, the + concurrency gate that finds a routine's active run by dispatch fingerprint + shall return exactly what it would have returned had no sweep run. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.4:** After any sequence of sweeps, the + lookup that closes out a routine run when its linked task reaches a terminal + step shall not resolve a deleted row to a different routine's run. A deleted + row resolves to nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.5:** Retention shall not change the + behavior of deleting a routine or deleting a workspace. Those paths continue + to remove every row they remove today, including rows retention has not yet + reached. + +## Out of scope + +Each exclusion below is a decision, not an oversight. + +- **Automation run history (`automation_runs` and its tables).** Owned by the + automation system, and its published contract is that history is removed only + by an explicit per-run or delete-all action. It has the same unbounded-growth + gap and needs its own requirement; changing it here would silently break a + documented promise. +- **`office_activity_log`, inbox dismissals, and approval rows.** The + [inbox requirement](inbox.md) already states these accumulate indefinitely and + excludes activity-log retention. Reopening it belongs to that contract. +- **`office_cost_events`.** Deleting cost rows changes reported spend and budget + enforcement: a decision about financial records, not about storage. +- **Detecting a stuck routine.** The 323-row reference case came from a routine + that fired correctly and did nothing useful for 28 days. Bounding its history + does not detect it; a detector for that state is a scheduler or stall + visibility concern. +- **Reclaiming file bytes after deletion.** Deleting rows does not shrink a + SQLite file. `VACUUM` is already an operator action on the System pages and a + sweep does not trigger it. +- **Archival or export before deletion.** Deleted history is gone. An install + needing it kept sets a longer window or takes a backup. +- **Per-workspace or per-routine retention overrides.** Settings are + instance-wide here. The retention floor is already per owner, which covers what + a per-routine override would most often be used for. +- **A persisted history of retention sweeps.** The last sweep's result is held + in memory and reported alongside the durable warnings. A growing table + recording the work of the job that stops tables growing is the same defect in + a new place. +- **Tables outside Office.** Nothing here changes task, session, workflow, + plugin, or auth storage. + +## Prior art + +Receipts for both legs. The design document carries what each finding changed. + +**Our own wiki: not consulted, tool unavailable.** The `@henry` pin resolved +`~/.obsidian-wiki/config` to `config.henry`, giving +`OBSIDIAN_VAULT_PATH=/Users/henry/Documents/henry/wiki` and +`QMD_WIKI_COLLECTION=wiki`. Neither retrieval path ran: `qmd` and +`obsidian-wiki` are absent from `PATH`, no QMD MCP tool is registered in this +session, and the vault directory returns `Operation not permitted` both +sandboxed and unsandboxed, a macOS file-access restriction on this process +rather than a missing vault. The grep fallback is blocked the same way, so this +leg is a tooling gap, not evidence the wiki is silent on retention. + +**What other products shipped: consulted.** Queried the `saas-kb` server +(`search_fsm_docs`, `category: "ai_sdlc"`) three times: run-history retention and +database growth; session history retention and automatic deletion; and +scheduled-automation run-history limits. Relevance was low across all three, +itself a finding about corpus coverage. Two useful hits: **GitLab Duo** deletes +sessions 30 days after last activity, which anchors the default window here; and +**the Claude apps gateway** documents four tables with per-table windows +enforced by one hourly sweep, marking one table "until deleted via the API" +instead of giving it a window, which is the split this document draws between +history rows and live-state rows. + +Neither previewed before a policy's first deletion, nor warned ahead of the +window. Those are where this capability goes further, because an upgrade that +silently deletes a year of history on its first sweep is the failure mode a +shipped default carries. diff --git a/docs/specs/office/system-design/run-history-retention-operations.md b/docs/specs/office/system-design/run-history-retention-operations.md new file mode 100644 index 00000000000..bcc04167250 --- /dev/null +++ b/docs/specs/office/system-design/run-history-retention-operations.md @@ -0,0 +1,376 @@ +--- +status: current +system: office +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 +--- + +# Office Run History Retention Operations System Design + +## Purpose and boundaries + +This design covers the operator half of run history retention: the settings +record, the per-table preview marker, the reporting value, the health check, and +the read/write System page surface. The sweep itself, the eligibility +predicates, the batching, and the engine-parity guarantees are designed in [run +history retention](run-history-retention.md). + +The two documents share one component, `internal/office/retention`, and one +scheduler goroutine. They are split because they are two contracts with two +consumers: a scheduler deleting rows, and an operator reading a page. The split +also keeps each document inside the specification size limit without cutting +either contract. + +Adjacent contracts read and constrained but not owned: + +- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health` + response the System page's health card renders. +- `internal/system/settings.Store` — the key/value settings table, already used + by storage maintenance under one JSON key with normalization on read. + +## Component: the operator surface + +### Settings + +One `system_settings` key, `office_run_retention`, holding a JSON document, +read through a `Get`/`Save` pair with `Normalize` on both, matching +`internal/system/storage/settings.go`. Unparseable content returns the defaults +plus a sentinel error the caller turns into a health issue, never a boot failure +(AC-OFFICE-RUN-HISTORY-RETENTION-004.4). + +```json +{ + "enabled": true, + "sweep_interval_hours": 6, + "batch_limit": 5000, + "routine_runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 }, + "runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 }, + "run_events": { "warn_rows": 250000 } +} +``` + +`run_events` carries only a threshold: its lifetime is its run's, so it has no +window and no floor of its own. Ranges and rejection behavior are +AC-OFFICE-RUN-HISTORY-RETENTION-004.2 and .3; validation returns a field-named +error and writes nothing, as `validateRange` does for storage maintenance. +Writes are last-writer-wins through `Store.Save`; `CompareAndSwap` is not used +because these are operator-scale settings edited from one page, and an identical +repeated write is indistinguishable from no write +(AC-OFFICE-RUN-HISTORY-RETENTION-004.5). + +**Each sweep re-reads the settings from the store at its start, and skips if that +read fails.** The buffered +wake channel that re-arms the interval is process-local, so on a deployment where +several backends share one PostgreSQL database — the deployment +AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists for — it only ever reaches the +process that served the `PUT`. A backend relying on a cached effective value +could win the sweep lock still holding the policy the operator has just replaced +and delete rows the new policy retains. Re-reading costs one indexed key lookup +per sweep interval, against the risk of deleting history under a window the +operator already widened. The wake channel keeps its job — re-arming the timer +promptly in the process that saw the change — and stops being the only path by +which a change reaches a sweep. + +A failed or unparseable re-read **skips the sweep** rather than falling back to +the documented defaults. Everywhere else in this design an unreadable settings +document yields the defaults and a health issue +(AC-OFFICE-RUN-HISTORY-RETENTION-004.4), which is right for reporting and for +startup because neither deletes anything. It is wrong here: the default window is +30 days, an operator who widened theirs to 3,650 has by definition configured +something longer, and falling back would delete the decade of history they +configured the system to keep. Reading settings to *show* them fails open; +reading them to *delete by* fails closed. + + +### The preview marker + +A second `system_settings` key, `office_run_retention_preview_completed`, +holding a JSON object keyed by swept table name whose values are the timestamp +at which that table's preview completed: +`{"office_routine_runs": "...", "runs": "..."}`. + +The marker is **per table, not per database**. A single global flag is unsafe +against AC-OFFICE-RUN-HISTORY-RETENTION-002.7, which lets one table fail while +the sweep as a whole still completes: if `runs` errored during the preview sweep +and `office_routine_runs` succeeded, a global flag would be written anyway and +`runs` would delete for real on the next sweep having never shown an operator a +would-delete count — the exact failure this capability exists to prevent. With a +per-table marker, `runs` is simply previewed again next sweep +(AC-OFFICE-RUN-HISTORY-RETENTION-003.4). + +A table's entry is written only when that table's preview evaluation completed +successfully, is never cleared by a settings change or a restart, and is not +written by a deleting sweep. A table whose preview finds nothing still gets its +entry, so its next sweep deletes normally. + +If the key is present but unparseable, every swept table is treated as **not yet +previewed** and `office_retention_preview_unreadable` is raised +(AC-OFFICE-RUN-HISTORY-RETENTION-003.10). The asymmetry is deliberate: a +spurious re-preview deletes nothing and costs one sweep, whereas assuming a +preview had completed permits a first deletion no operator ever saw. The safe +direction is the one that cannot delete. + +The preview's per-table counts are uncapped by the batch limit +(AC-OFFICE-RUN-HISTORY-RETENTION-003.9): capping them would report 5,000 to an +operator holding 200,000 eligible rows, which is exactly the number the warning +exists to convey. + + +### Reporting + +An in-memory `LastSweep` value replaced wholesale at the end of each sweep that +actually ran: start and finish times, and per **reported** table the deleted +count, a backlog flag, and an error string. Deleted counts are counted from +**committed** transactions only: when a `runs` batch is abandoned and rolled back +(AC-OFFICE-RUN-HISTORY-RETENTION-002.7), the three satellite tables whose rows +that transaction addressed report **zero** deleted for that sweep, not the counts +their statements returned before the rollback. The failure is recorded against +`runs`, but the satellites must not show rows the rollback restored — the one +surface an operator has for "what actually happened" would otherwise be wrong +precisely in the failure case it exists for. Per **swept** table it also carries +whether that table was previewed in this sweep and, when it was, its +`WouldDelete` count — without that field the preview's headline numbers, which +AC-OFFICE-RUN-HISTORY-RETENTION-003.2 and +AC-OFFICE-RUN-HISTORY-RETENTION-003.9 require an operator to see, would have +nowhere to be read from. Because the preview marker is per table, `preview` is a +per-table flag rather than one flag for the sweep. Skips are held separately and +never overwrite this value. Nothing is persisted +(AC-OFFICE-RUN-HISTORY-RETENTION-004.6, and the "no persisted sweep history" +exclusion: a growing table recording the work of the job that stops tables +growing is the same defect in a new place). Before the first sweep the value is +absent, and the surface says so rather than rendering zeros +(AC-OFFICE-RUN-HISTORY-RETENTION-004.7). + +**Retained counts do not come from `LastSweep`.** They are a property of the +table, not of a sweep, and three ACs need them when no sweep has run at all: +AC-OFFICE-RUN-HISTORY-RETENTION-003.7 (threshold warning while retention is +disabled), AC-OFFICE-RUN-HISTORY-RETENTION-004.8 (surface readable while +disabled), and AC-OFFICE-RUN-HISTORY-RETENTION-002.8 (disabled means no sweep, +yet counts are still reported). A separate `RetainedCounts` value therefore +holds one count per thresholded table — produced by the census described next — +plus, for `office_routine_runs`, the top routine by retained rows for +AC-OFFICE-RUN-HISTORY-RETENTION-003.5's attribution. + +**The count evaluation is a status census, not a bare `COUNT(*)`.** For each +swept table it issues one +`SELECT status, COUNT(*) FROM
GROUP BY status`. The retained count is the +sum of those rows, so the census costs one scan rather than two, and the same +result set is what detects a status in neither the history nor the live-state set +and raises `office_retention_unknown_status:
` +(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). That detector has to live here rather +than in the sweep: the sweep's predicate selects `status IN ()` +and therefore structurally cannot observe a status it does not select. +`run_events` is thresholded but not swept and has no status column, so it keeps a +plain `COUNT(*)`. + +It is refreshed on the sweep interval by the same scheduler goroutine — on that +schedule whether or not retention is enabled, since a disabled install is +precisely the one whose tables grow unattended, and is also the only install +where an unrecognized status would otherwise never be noticed — and served from +memory in between. + +**The first evaluation runs at `Start`, not at the first sweep.** The first sweep +is deliberately delayed five minutes (AC-OFFICE-RUN-HISTORY-RETENTION-002.10); +hanging the first count off that tick would leave every restart with five minutes +of absent counts, and a fresh install with none at all until it had swept once — +while AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and -004.8 require the threshold +warning and the surface to work on exactly those installs. + +It runs **on the scheduler goroutine, not on the caller of `Start`**. The census +is three unbounded scans of the largest tables in the database, and the install +that most needs them is the one where they take longest; blocking `Start` on them +would make bounding these tables a cause of slow boots. `Start` returns +immediately, the tables read *not yet computed* until the first census lands +seconds later, and that state is already required and already renders honestly. + +`RetainedCounts` is therefore **tri-state per thresholded table**, not a number +that defaults to zero: *not yet computed*, *fresh as of T*, or *stale as of T*. +Zero is a real measurement and must not be how "nobody has counted yet" renders — +the same argument AC-OFFICE-RUN-HISTORY-RETENTION-004.7 makes for `LastSweep`, +and the reason AC-OFFICE-RUN-HISTORY-RETENTION-003.11 states it. The three states +are tracked **per table**, so one table's failing query neither discards nor +staleness-marks another's successful one; a failure with no prior success leaves +that table at *not yet computed* rather than fabricating a zero. + +The evaluation is explicitly **not** computed per health poll or per page load: +`/api/v1/system/health` is polled by every open browser tab, and issuing three +unbounded `COUNT(*)`s against the largest tables in the database on each poll +would make this feature a cause of the load it exists to prevent +(AC-OFFICE-RUN-HISTORY-RETENTION-003.11). + +Three surfaces, in descending durability: + +1. **Structured logs.** One `info` line per sweep. One `warn` line per condition + in REQ-OFFICE-RUN-HISTORY-RETENTION-003, carrying the table, the counts, and + the threshold. Always on, in every profile. +2. **Health issues.** The package implements `health.Checker` with + `Name() = "Office run retention"` and `Category() = "office"`, returning a + `health.Issue` per active condition with + `FixURL = "/settings/system/data-storage"`, which is the live route + registered in `apps/web/src/settings-routes.tsx` — note the suffix, as + `/settings/system/data` is not a registered path and `/settings/system/database` + is only a redirect to it. `internal/health/checks_test.go` pins fix URLs to + live routes ("expectedGitHubFixURL is the live route in ..."), and this one + is pinned the same way. This is the production-visible surface + required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 and is why the debug + metrics endpoint is not it: `/debug/vars` is gated on the dev profile. + Issue ids are stable and one per condition: + `office_retention_preview_pending`, `office_retention_backlog:
`, + `office_retention_threshold:
`, `office_retention_disabled:
`, + `office_retention_settings_invalid`, `office_retention_failed:
`, + `office_retention_preview_unreadable`, and + `office_retention_unknown_status:
`. + `Check` returns issues sorted by id, matching `storage.Runtime.Check`, so the + health card's order is stable across polls. + For `office_routine_runs`, the threshold issue's message names the routine + holding the largest share of retained rows and that share + (AC-OFFICE-RUN-HISTORY-RETENTION-003.5) — attribution computed from the same + retained-count query, not a second detector. +3. **expvar.** Counters under `office_retention_*` following + `office/scheduler/metrics_vars.go`. Development convenience only; nothing in + REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on it. + + +### HTTP and frontend + +`GET /api/v1/system/retention` returns effective settings, `LastSweep`, the +separately-held skip record, and `RetainedCounts`. + +`PUT /api/v1/system/retention` **replaces the whole document**; it is not a +merge patch. A field the caller omits takes its documented default rather than +its stored value, so the same request body always yields the same stored +document and a repeated identical write is genuinely a no-op — which is what +AC-OFFICE-RUN-HISTORY-RETENTION-004.5's "identical repeated write changes +nothing and returns the same normalized document" requires. A merge would break +that: under a merge, whether a request is a no-op depends on what was stored +before it. An unrecognized field, a numeric field whose value is not a whole +number, a field present with a JSON `null`, and a field whose value is of the +wrong type, are each rejected with a field-named 400 and nothing is written +(AC-OFFICE-RUN-HISTORY-RETENTION-004.9); rejecting rather than ignoring an +unknown field means a client that misspells `window_days` is told so instead of +silently getting the default. `null` needs saying because this endpoint is a full +replace: since an omitted field deliberately means "take the default", the +tempting reading of `null` is the same one, and that reading turns +`{"runs": {"window_days": null}}` into a silent 3,650-to-30-day reduction that +destroys a decade of history on the next sweep. Decode into pointer fields and +reject an explicit null, rather than into value fields where `null` and omission +are indistinguishable. Successful writes return the normalized document. + +Both routes are admin-scoped like the other System routes and are readable while +retention is disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8). + +One card on **Settings > System > Data & Logs**, the page served at +`/settings/system/data-storage` and rendered by +`apps/web/components/settings/system/data-logs-settings.tsx`, beside +`database-stats-card.tsx`: the enable toggle, the numeric fields, and the +last-sweep readout. It follows the storage-maintenance cards' shape. All new +copy goes through `t()` and must ship in `pt-pt`, `zh-cn`, `zh-hk`, and `zh-tw`; +`pnpm run i18n:check` and the new-code ratchet gate the build. Health issue +titles and messages stay English, matching every other backend-produced +`health.Issue`. + + +## Ordering, concurrency, and failure + +| Question | Answer | AC | +|---|---|---| +| Concurrent settings writes | Last-writer-wins; an identical repeat changes nothing | 004.5 | +| Settings unreadable | Defaults used, health issue raised, sweep proceeds | 004.4 | +| Settings out of range | Rejected with the field named, stored settings unchanged | 004.3 | +| Retention disabled | No sweep, no deletion; counts and threshold warnings still reported | 002.8, 003.7, 004.8 | +| Preview scope | Per swept table, not per database; a table that failed its preview is previewed again | 003.4 | +| Preview marker unreadable | Treated as not previewed; re-preview deletes nothing | 003.10 | +| Retained counts | Table property from a `GROUP BY status` census; first evaluation at `Start`, then on the sweep interval; served from memory; computed even while disabled | 003.11 | +| Counts before the first evaluation, or first evaluation fails | Reported *not yet computed* per table, never as zero; state tracked per thresholded table | 003.11, 004.7 | +| Unknown status detected | By the census, not the sweep predicate; one issue per table listing every unrecognized status in ascending order | 001.10 | +| Preview on a table with more eligible rows than the batch limit | Reports the full eligible count; never reported as backlog | 003.6, 003.9 | +| Settings changed on another backend | Each sweep re-reads settings at its start, so a process that did not serve the write still uses the new policy | 004.5 | +| Settings unreadable at sweep start | Sweep skipped and recorded as skipped; never run under the shorter default window | 004.5, 004.4 | +| Batch abandoned and rolled back | Satellite tables report zero deleted, not their pre-rollback statement counts | 002.7, 004.6 | +| Settings write shape | Full replace; omitted field takes its default; unknown field 400 | 004.9 | + +## Testing + +Unit tests in `internal/office/retention`, plus the frontend checks below. + +- The first sweep on a seeded database deletes nothing and reports a + would-delete count; the second deletes (003.1, 003.4). +- A preview on a database with more eligible rows than the batch limit reports + the full eligible count (003.9). +- A table that errors during the sweep in which it was being previewed is + previewed again on the next sweep instead of deleting, while a sibling table + that succeeded proceeds to delete (003.4). +- A preview marker that is present but unparseable causes a re-preview, not a + deletion, and raises `office_retention_preview_unreadable` (003.10). +- Retained counts and the threshold warning are produced on a fresh install + before any sweep, and while retention is disabled (003.7, 003.11, 004.8), and + a health poll does not issue a table count. +- Counts are available immediately after `Start`, without advancing any clock to + the first sweep's delay (003.11). A test that waits out the delay would pass + against an implementation that hangs the first count off the sweep tick, which + is the defect this asserts against. +- Before the first evaluation, and when the first evaluation fails with no + earlier success, the surface reports the table as *not yet computed* and not as + zero (003.11, 004.7). +- With one thresholded table's count query failing and the other two succeeding, + the two keep fresh counts and only the failing one is marked, and the threshold + warnings derived from the successful counts still fire (003.11). +- A swept table holding a status in neither status set raises + `office_retention_unknown_status:
` from the census while retention is + **disabled** and no sweep has ever run (001.10, 003.11). +- A preview on a table with more eligible rows than the batch limit reports the + full eligible count and raises no backlog warning (003.6, 003.9). +- A settings change written through one store handle is used by a sweep driven + from a second handle that never saw the change notification (004.5). +- A sweep whose settings read fails is skipped and recorded as skipped, and + deletes nothing under the default window (004.5). +- A `runs` batch abandoned after its retry reports zero deleted for the three + satellite tables rather than their pre-rollback counts (002.7, 004.6). +- A recorded skip leaves the previous `LastSweep` readable and unchanged (002.2, + 004.7). +- A settings write omitting a field stores that field's default; a write with an + unknown field or a fractional number is rejected 400 naming the field and + stores nothing; the same write applied twice is a no-op (004.9). + +Commands: + +``` +cd apps/backend +go test ./internal/office/... -race -count=1 +KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... -count=1 +cd apps/web && pnpm run typecheck && pnpm run i18n:check +``` + +## Rejected alternatives + +- **One global "preview completed" flag.** Simpler, and unsafe: because + AC-OFFICE-RUN-HISTORY-RETENTION-002.7 lets one table fail while the sweep + completes, a global flag is written even when a table never got its preview, + and that table then deletes for real having shown the operator nothing. The + per-table marker costs one JSON object. +- **Compute retained counts inside the health check.** Direct, and it puts three + unbounded `COUNT(*)`s on a route every open browser tab polls. Refreshing on + the sweep interval and serving from memory gives the same number without + making the bound-the-tables feature a source of load on those tables. +- **A merge-patch settings write.** Under a merge, whether a request is a no-op + depends on what was stored before it, which + AC-OFFICE-RUN-HISTORY-RETENTION-004.5 forbids. +- **Deletion off by default.** Safe, and it means the gap stays open on every + install that never visits the settings page. The per-table preview gives the + same protection without that outcome. +- **Report warnings only through `/debug/vars`.** That endpoint is gated on the + dev profile, so on a production build the warnings would not exist + (AC-OFFICE-RUN-HISTORY-RETENTION-003.8). + +## Prior art, applied + +`internal/system/storage` supplies the settings storage and normalization +pattern, the hours-based interval with min and max bounds, and the +`health.Checker` route to a production-visible warning; `storage.Runtime.Check` +also sorts its issues by id, which this check matches so the health card's order +is stable across polls. + +Neither GitLab Duo nor the Claude apps gateway previews before a policy's first +deletion or warns ahead of the window. Those are this capability's additions, +and they are the whole reason this document is separate from the sweep's. diff --git a/docs/specs/office/system-design/run-history-retention.md b/docs/specs/office/system-design/run-history-retention.md new file mode 100644 index 00000000000..31868ae5def --- /dev/null +++ b/docs/specs/office/system-design/run-history-retention.md @@ -0,0 +1,557 @@ +--- +status: current +system: office +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +--- + +# Office Run History Retention System Design + +## Purpose and boundaries + +This design adds one scheduled sweep that deletes aged Office run history from +five tables. It changes no run lifecycle, no routine dispatch decision, and no +task, session, or checkout. + +Office owns the sweep because the eligibility rules are defined by Office +primitives. The settings record, the health check, the reporting value, and the +System page surface are the operator half of the same capability and are +designed in [run history retention +operations](run-history-retention-operations.md), which is where every +requirement in REQ-OFFICE-RUN-HISTORY-RETENTION-003 and -004 is satisfied. + +Adjacent contracts read and constrained but not owned: + +- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health` + response the System page's health card renders. +- `internal/system/settings.Store` — the key/value settings table, already used + by storage maintenance under one JSON key with normalization on read. +- `internal/runs/repository/sqlite` — the `runs` queue and `run_events` access + methods. +- `internal/office/repository/sqlite` — the schema owner for `runs`, + `run_events`, `office_run_route_attempts`, `office_run_skills`, and + `office_routine_runs`. + +## Measured starting state + +Read from the reference install's SQLite database on 2026-09-09. These numbers +set the defaults and are the baseline any regression test can be written +against. + +| Table | Rows | Notes | +|---|---|---| +| `office_routine_runs` | 326 | 323 `coalesced`, 3 `task_created` | +| `runs` | 53 | | +| `run_events` | 340 | across 53 runs, mean 6.4 per run | +| `office_run_route_attempts` | 55 | | +| `office_run_skills` | 382 | | + +The 323 `coalesced` rows span 2026-08-03 to 2026-08-31 and belong to a single +routine that is now `paused`. Orphan `run_events` today: zero, because nothing +has ever deleted a `runs` row. + +## Schema facts this design depends on + +Verified by reading the schema owners, not assumed. + +- `office_routine_runs` (`office/repository/sqlite/base.go`) has + `FOREIGN KEY (routine_id) REFERENCES office_routines(id) ON DELETE CASCADE` + on both engines, and SQLite opens with `_foreign_keys=on` + (`internal/db/sqlite.go`). Its two indexes are partial and serve the dispatch + gate, not an age scan: `idx_office_routine_runs_active_fingerprint` + (`WHERE status = 'task_created'`) and `idx_office_routine_runs_linked_task` + (`WHERE linked_task_id != ''`). +- `run_events`, `office_run_route_attempts`, and `office_run_skills` declare + **no foreign key to `runs`** on either engine. Confirmed in + `office/repository/sqlite/base.go` and in the PostgreSQL conformance snapshot + `internal/persistence/storeconformance/testdata/upgrades/v0.93.0/postgres.sql`, + where `run_events` has only a primary key and one index. A cascade would + therefore delete nothing; satellite deletion must be explicit + (AC-OFFICE-RUN-HISTORY-RETENTION-005.2). +- `run_events.seq` is assigned by `AppendRunEvent` as + `COALESCE(MAX(seq) + 1, 0)` scoped to the run. Deleting a live run's whole + timeline restarts its sequence at zero, and the run detail view's incremental + tail reads `WHERE seq > afterSeq`. This is why + AC-OFFICE-RUN-HISTORY-RETENTION-001.6 forbids event deletion outside the + transaction that deletes the run. +- `runs` indexes are `idx_run_status_requested (status, requested_at)` and the + partial unique `idx_run_idempotency`. Nothing indexes `finished_at`. +- `ScheduleRetry` (`runs/repository/sqlite/runs.go`) sets + `status = 'queued', finished_at = NULL` on an existing run, keyed by id with + **no status guard in its `WHERE` clause**. Any terminal run can therefore + become live again at any moment, which is the race + AC-OFFICE-RUN-HISTORY-RETENTION-002.4 closes. Because the guard is absent, a + `cancelled` row is resurrectible on exactly the same terms as a `failed` one, + so admitting `cancelled` to the history set adds no new race — it is covered + by the same re-assertion. +- `CancelRunsWhere` (`runs/repository/sqlite/cancel.go`) is documented as "the + single writer of the terminal cancel state on the runs table". It sets + `status = 'cancelled', cancel_reason = ?, finished_at = ?` and is guarded by + `status IN ('queued', 'claimed')`, so a cancelled row is always terminal and + always carries a completion timestamp. It is production-reachable through + `CancelRunsForTasks` from `office/service/tree_controls.go` and + `office/repository/sqlite/participants.go`. It writes the literal string, so + `office/models/enums.go`'s four-value `RunStatus` block does not enumerate it: + the database has five `runs` statuses, not four. This is why + AC-OFFICE-RUN-HISTORY-RETENTION-001.2 classifies `cancelled` as history and + AC-OFFICE-RUN-HISTORY-RETENTION-001.10 makes any sixth value fail safe. +- Every production writer of a terminal `runs` status stamps the completion + timestamp: `FinishRun` sets `finished_at = now`, `MarkRunFailed` sets + `finished_at = COALESCE(finished_at, now)`, and `CancelRunsWhere` sets it + outright. Only the test helper `SetRunStatusForTest` can leave a terminal row + with a null `finished_at`. The `COALESCE(finished_at, requested_at)` fallback + in AC-OFFICE-RUN-HISTORY-RETENTION-001.3 is therefore defensive rather than a + routine path — but it is also what makes the ordering key non-null, which is + what keeps the floor deterministic across engines (see below). +- `runs.requested_at` and `office_routine_runs.created_at` are both + `TIMESTAMP NOT NULL`, so the completion-time expression is total. This matters + for parity, not just tidiness: SQLite and PostgreSQL differ in where they sort + nulls by default, so an `ORDER BY` over a nullable timestamp would rank the + floor differently on the two engines from identical data. +- `CleanExpired` (`runs/repository/sqlite/runs.go`) already deletes terminal + `runs` rows older than a cutoff and **has no production caller** — only + tests reach it. It deletes only `runs`, so wiring it as-is would orphan every + satellite row. It is superseded by the batched, satellite-aware delete below + rather than reused. + +## Component: `internal/office/retention` + +One package holding policy, settings, the sweep, and the health check. + +### Ownership and lifecycle + +A single goroutine owner modelled on `internal/system/storage.Scheduler`: a +`Start(ctx)` that is a no-op when already running, a `Stop()` that cancels and +joins, and a buffered wake channel so a settings change re-arms the interval +without interrupting a sweep in progress +(AC-OFFICE-RUN-HISTORY-RETENTION-004.5). `Stop` is joined from the same place +that stops the Office scheduler. + +The loop is deliberately not the Office run-processing tick +(`office/service/scheduler_integration.go`, `DefaultTickInterval = 5s`). That +tick already carries `RecoverStale` and `ReapStaleCheckouts` unthrottled and is +the run-claim hot path; on SQLite a bulk delete there contends with the single +writer that claims runs, 17,280 times a day, for an input that changes on a +scale of days (AC-OFFICE-RUN-HISTORY-RETENTION-002.1). + +A `sweeping bool` guarded by the same mutex makes a due sweep a skip rather than +a second goroutine (AC-OFFICE-RUN-HISTORY-RETENTION-002.2). That guard is +process-local, which is sufficient on SQLite, where the database file is owned +by one backend process. On PostgreSQL, where several backends can share one +database, it is not: two backends would each see `sweeping == false` and sweep +the same rows concurrently. + +The sweep therefore also takes a PostgreSQL advisory lock, and it must be a +**session-scoped, non-blocking** one, which is a different shape from every +existing advisory lock in this repository: + +``` +conn := db.Conn(ctx) // one dedicated connection +SELECT pg_try_advisory_lock(:retention_key) // boolean, returns immediately +... whole sweep, every table, every batch, on the pool ... +SELECT pg_advisory_unlock(:retention_key) // on that same connection +conn.Close() // deferred +``` + +Both properties are load-bearing and neither is optional: + +- **Session-scoped, not transaction-scoped.** The sweep is multi-transaction by + construction: AC-OFFICE-RUN-HISTORY-RETENTION-002.5 makes each batch its own + transaction, and AC-OFFICE-RUN-HISTORY-RETENTION-002.7 requires one table's + failure to leave another table's committed deletions intact, which forbids + wrapping the sweep in a single transaction. A `pg_advisory_xact_lock` is + released when its transaction ends, so it would protect one batch and then let + a second backend in between tables — the interleaving + AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent. The lock is held on a + connection checked out for the sweep and released in a `defer`; the batches + themselves continue to use the pool. +- **`try`, not the blocking form.** AC-OFFICE-RUN-HISTORY-RETENTION-002.12 says a + backend that cannot acquire the lock *skips*. `pg_advisory_xact_lock` and + `pg_advisory_lock` wait instead of failing, which would convert a concurrent + sweep into a queued one and eventually stall the scheduler behind a long sweep. + `pg_try_advisory_lock` returns `false` immediately; on `false` the backend + records a skip and returns, exactly as for a local concurrent sweep. + +This deliberately departs from the established repo pattern +`SELECT pg_advisory_xact_lock(hashtextextended(?, 0))` +(`office/repository/sqlite/participants.go`, `internal/secrets/sqlite_store.go`, +`internal/workflow/repository/phase2_sqlite.go`), and the departure is the point: +each of those call sites performs its entire unit of work inside the one +transaction that holds the lock, and each *wants* to wait rather than skip. The +sweep can do neither. The key is derived the same way, `hashtextextended` over a +constant distinct from every key those sites use, so retention never contends +with participant-seat, secret-transfer, or workflow-phase locking. + +**If the lock connection drops mid-sweep**, PostgreSQL releases the session's +advisory locks as part of ending the session, so a crashed or partitioned backend +cannot wedge retention permanently — that self-healing is the reason for a +session lock rather than a lease row in `system_settings`, which would need its +own expiry and its own stale-holder rule. The cost is that the surviving sweep no +longer holds exclusivity without knowing it, so the sweep **verifies the lock +connection is still alive between tables** and, if it is not, stops before the +next table, records that table as skipped rather than failed, and returns. It +does not attempt to re-acquire mid-sweep: a re-acquisition after another backend +has taken the lock would produce exactly the concurrent sweep this protects +against. Batches already committed stay committed, which +AC-OFFICE-RUN-HISTORY-RETENTION-002.5 permits. + +A skip is recorded as its own value — a skip counter and a last-skip timestamp — +and does **not** overwrite `LastSweep`. `LastSweep` holds the last sweep that +actually ran, so a burst of skips cannot blank the operator's view of the last +real result (AC-OFFICE-RUN-HISTORY-RETENTION-002.2, and +AC-OFFICE-RUN-HISTORY-RETENTION-004.7, which forbids rendering a +never-swept-looking surface when a sweep has in fact run). + +The first sweep after `Start` is armed at a fixed 5-minute delay rather than a +full interval (AC-OFFICE-RUN-HISTORY-RETENTION-002.10). Arming at the interval, +as the storage scheduler does with its 24-hour default, means an install +restarted more often than the interval never sweeps at all; 5 minutes keeps +startup clear of schema init and run recovery without depending on uptime. Each +sweep computes `time.Now().UTC()` once and passes that instant to every table, +so two tables in one sweep cannot disagree about the cutoff +(AC-OFFICE-RUN-HISTORY-RETENTION-002.11). + +Scheduling is **fixed-delay, not fixed-rate**: the next sweep is armed when the +previous one returns, so a sweep that overruns its interval delays its successor +instead of causing an immediate second one (which the concurrency guard would +only skip anyway, turning a slow sweep into a stream of skips). A settings +change re-arms the delay from the moment of the change rather than from the last +sweep, and enabling retention that was disabled arms at the same 5-minute delay +as a fresh start rather than at a full interval — otherwise an operator who +enables retention on a 168-hour interval waits a week to see whether it works +(AC-OFFICE-RUN-HISTORY-RETENTION-002.13). + +### Eligibility, expressed once + +Two predicates, each defined in exactly one place and reused by the count, the +preview, and the delete. + +**Routine runs.** History statuses are `skipped`, `coalesced`, `failed`, `done`, +`cancelled`. `received` and `task_created` are absent by construction, which is +what makes AC-OFFICE-RUN-HISTORY-RETENTION-005.3 hold: the dispatch gate +`GetActiveRunForFingerprint` reads only `status = 'task_created'`, so no sweep +can change its answer. + +``` +DELETE FROM office_routine_runs +WHERE id IN ( + SELECT id FROM ( + SELECT id, + ROW_NUMBER() OVER ( + PARTITION BY routine_id + ORDER BY COALESCE(completed_at, created_at) DESC, id DESC + ) AS rn + FROM office_routine_runs + WHERE status IN () + ) ranked + WHERE rn > :floor + AND completion_time < :cutoff + ORDER BY completion_time ASC, id ASC + LIMIT :batch +) +AND status IN () +AND COALESCE(completed_at, created_at) < :cutoff +``` + +where the ranked subquery also projects +`COALESCE(completed_at, created_at) AS completion_time`. + +Two orderings appear here and they are not the same ordering; conflating them is +the defect this section exists to prevent. + +- The **`ORDER BY` inside `ROW_NUMBER()`** ranks rows *within* a partition so the + floor keeps the newest per owner: `completion_time DESC, id DESC` + (AC-OFFICE-RUN-HISTORY-RETENTION-001.4). +- The **`ORDER BY` on the outer select** decides *which* eligible rows a + batch-limited sweep takes: `completion_time ASC, id ASC`, oldest first + (AC-OFFICE-RUN-HISTORY-RETENTION-002.3). The window function's ordering does + not reach the outer `LIMIT`, so without this clause the engine is free to + return any subset and the two engines may drain a backlog differently from + identical data — which would contradict + AC-OFFICE-RUN-HISTORY-RETENTION-005.1 and make the parity test below flake for + a reason unrelated to a genuine engine difference. + +`id` is a UUID and so is not chronological; it is used only as a total, stable +tiebreak for equal timestamps, in both orderings. + +The trailing `AND` clauses are the re-assertion required by +AC-OFFICE-RUN-HISTORY-RETENTION-002.4. For this table the whole ranked subquery +is re-evaluated inside the `DELETE`, so the floor is re-asserted atomically along +with status and age; the two-phase `runs` path below has to do that explicitly. + +Window functions are available on both engines: SQLite 3.54.0 through +`github.com/mattn/go-sqlite3 v1.14.33`, verified by running this exact +`ROW_NUMBER() OVER (PARTITION BY routine_id ...)` against the reference +database. + +**Runs.** History is `status IN ('finished','failed','cancelled')`, partitioned +by `agent_profile_id`, ranked `COALESCE(finished_at, requested_at) DESC, id DESC` +and batch-ordered `COALESCE(finished_at, requested_at) ASC, id ASC`. There is no +`finished_at IS NOT NULL` conjunct: requiring one would make a terminal row with +an unset stamp immortal and unobservable, which +AC-OFFICE-RUN-HISTORY-RETENTION-001.3 forbids. `queued` and `claimed` are +absent, which covers a routing-parked run whose `earliest_retry_at` is far in +the future (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). + +A status in neither set is treated as live state and raises +`office_retention_unknown_status:
` +(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). Both sets are closed and asserted +against `office/models/enums.go` plus the literal-SQL writers in a test, so a +sixth status cannot enter the database without failing that test — the failure +mode that hid `cancelled` in the first place. + +**That warning needs a producer, and the eligibility predicate cannot be it.** +The predicate selects `status IN ()`, so a row holding an +unrecognized status is never selected, never counted, and never seen: a fail-safe +whose only detector is a CI test fires on the developer's machine and stays +silent on the install that actually has the row. The producer is instead the +**status census** — one + +``` +SELECT status, COUNT(*) FROM GROUP BY status +``` + +per swept table, run by the retained-count evaluation described in [run history +retention operations](run-history-retention-operations.md#reporting). Three +consequences follow from siting it there rather than in the sweep, and each one +closes a hole: + +- The census **subsumes the retained count** rather than adding a second scan: + the retained count is the sum of the census rows, so one `GROUP BY` yields + both. (`run_events` is thresholded but not swept and has no status column, so + it keeps a plain `COUNT(*)`.) +- It runs **whether or not retention is enabled**, because the count evaluation + does (AC-OFFICE-RUN-HISTORY-RETENTION-003.11). A disabled install is precisely + where an unrecognized status would otherwise never be noticed, since + AC-OFFICE-RUN-HISTORY-RETENTION-002.8 means no sweep runs there at all. +- It sees a status **because the row exists**, not because the row was + selectable, which is the property the eligibility predicate structurally + cannot have. + +One issue per table, not one per status: a table with several unrecognized +statuses raises the single id `office_retention_unknown_status:
` whose +message lists every unrecognized status **in ascending lexicographic order** with +its row count. Ordering is named because the message is compared across health +polls; an unordered list would make a stable condition look like a changing one. + +### Deleting a run + +Per batch, one transaction, satellites first, run last: + +1. Select up to `batch_limit` eligible run ids. +2. `DELETE FROM run_events WHERE run_id IN (...)` +3. `DELETE FROM office_run_route_attempts WHERE run_id IN (...)` +4. `DELETE FROM office_run_skills WHERE run_id IN (...)` +5. `DELETE FROM runs WHERE id IN (:selected_ids) AND id IN ()` — the re-assertion is the **whole ranked subquery**, not just the + status and cutoff conjuncts, so the retention floor is re-evaluated at delete + time along with them. Unlike the single-statement `office_routine_runs` + delete, this path selected its ids in a separate earlier statement, so a + `ScheduleRetry` in between can re-rank a partition and push a row that was + `rn > floor` at selection to `rn <= floor` now. Re-asserting only status and + age would delete a row that has since become floor-protected + (AC-OFFICE-RUN-HISTORY-RETENTION-002.4). + +Step 5 can delete fewer rows than steps 2 to 4 addressed, when a `ScheduleRetry` +resurrected a run between selection and delete. That is the correct outcome for +AC-OFFICE-RUN-HISTORY-RETENTION-002.4 only if the whole batch rolls back rather +than leaving a live run without its timeline. **The transaction is therefore +rolled back and retried once with a fresh selection when the step 5 row count +does not match the selected id count**; a second mismatch rolls back again and +abandons the batch. + +An abandoned batch is recorded as a **failure** for that table, raising +`office_retention_failed:
`, not as backlog +(AC-OFFICE-RUN-HISTORY-RETENTION-002.7). The distinction is not cosmetic: +backlog means work correctly deferred by the batch limit and is expected on a +large install, so routing this case there would file the one genuinely dangerous +outcome under the one routine one. The table is retried on the next scheduled +sweep either way, but only the failure classification tells an operator that a +deletion could not be applied consistently. + +Step 5 can only ever delete *fewer* rows than steps 2 to 4 addressed, never +more, because it is bounded by the same selected id set. This is the one place +where the naive implementation silently corrupts a live run, and it is the +reason satellite deletion is not a separate statement outside the transaction. + +`run_events` is never addressed by any predicate other than membership in this +id set (AC-OFFICE-RUN-HISTORY-RETENTION-001.6). There is no age-based delete on +`run_events`. + +### Indexes to add + +Neither table has an index serving the sweep. + +- `idx_office_routine_runs_retention ON office_routine_runs(routine_id, completed_at DESC, id DESC)` +- `idx_runs_retention ON runs(agent_profile_id, finished_at DESC, id DESC)` + +Added through the existing `office/repository/sqlite` schema path so both the +fresh-install `CREATE` and the upgrade path get them, and recorded in the +conformance fixtures. + +## Ordering, concurrency, and failure + +| Question | Answer | AC | +|---|---|---| +| Sweep ordering across tables | `office_routine_runs`, then `runs` with its satellites. Independent sets; the order is fixed only so results and logs are reproducible. | 002.6 | +| Floor ordering | `COALESCE(completed_at, created_at) DESC, id DESC` / `COALESCE(finished_at, requested_at) DESC, id DESC` | 001.4 | +| Two sweeps due at once | Second is skipped, not queued, and recorded as skipped | 002.2 | +| Sweep vs. live writer | Selection is a snapshot; status, age **and floor** are re-asserted in the delete; a batch whose step 5 count disagrees is rolled back | 002.4, 002.5 | +| Sweep interrupted | Each batch is one transaction; applied batches are whole, the interrupted one is not applied | 002.5 | +| First sweep after startup | Armed at a fixed 5-minute delay, not a full interval | 002.10 | +| Cutoff instant | Computed once per sweep, shared by every table | 002.11 | +| Re-run once no eligible rows remain | Deletes nothing further, reports zero. While backlog remains, the next sweep deletes the next batch by 002.3 — the two are not in tension because 002.6 is scoped to the drained state | 002.6, 002.3 | +| One table errors | That table is recorded failed; remaining tables continue; retried next sweep; startup unaffected | 002.7 | +| Empty tables | Zero deletions, no error, no warning | 002.9 | +| Deleted coalesce target | Allowed; `coalesced_into_run_id` is provenance, read by nothing | 001.9 | +| Routine paused | Irrelevant to eligibility | 001.7 | +| Cancelled run | History, like `finished`/`failed`; its writer stamps the completion timestamp | 001.2 | +| Terminal row, null completion stamp | Still history; dated by `created_at` / `requested_at` | 001.3 | +| Status in neither set | Treated as live state, never deleted, warned | 001.10 | +| Which rows a batch-limited sweep takes | `completion_time ASC, id ASC` — oldest first, named columns | 002.3, 005.1 | +| Batch abandoned after retry | Recorded as that table's failure, not as backlog | 002.7 | +| Preview finds more eligible rows than the batch limit | Not backlog. A preview deletes nothing by design, so "retention is behind" would be false; it reports the full eligible count instead | 002.3, 003.6, 003.9 | +| Sweep skipped | Recorded separately; never overwrites the last real `LastSweep` | 002.2, 004.7 | +| Two backends, one PostgreSQL | Session-scoped `pg_try_advisory_lock` held on a dedicated connection for the whole sweep; the loser skips without waiting | 002.12 | +| Lock connection drops mid-sweep | PostgreSQL releases the lock with the session; the sweep stops before the next table and records it skipped, and does not re-acquire | 002.12 | +| Scheduling model | Fixed-delay from the end of the previous sweep; a settings change re-arms from the change | 002.13 | + +## Testing + +Unit and repository tests in `internal/office/retention` and +`internal/office/repository/sqlite`, plus the persistence gates. + +Behaviors that must have a test, because each is a way the naive implementation +is wrong: + +- A `task_created` routine run older than any window survives, and + `GetActiveRunForFingerprint` still finds it (001.1, 005.3). +- A `queued` run with `finished_at` cleared by `ScheduleRetry` survives (001.2). +- A run resurrected between selection and delete leaves both the run and its + full `run_events` timeline intact (002.4, and the rollback above). +- After a sweep, no `run_events`, `office_run_route_attempts`, or + `office_run_skills` row references a missing `runs` row (001.5, 005.2). +- No `run_events` row of a surviving run is ever deleted (001.6). +- A routine with 3 history rows all older than the window keeps all 3 under a + floor of 50; a routine with 200 keeps exactly 50 (001.4). + +- Two sweeps triggered concurrently produce one sweep and one recorded skip + (002.2). +- A sweep hitting the batch limit reports backlog and the next sweep continues, + and its satellite rows are deleted in full rather than capped (002.3, 003.6). + +- A terminal row whose completion timestamp is unset is dated by `created_at` + or `requested_at` and ages out rather than being retained forever (001.3). + This test is only satisfiable because AC-OFFICE-RUN-HISTORY-RETENTION-001.2 + classifies history by status alone; an implementation that also required + `finished_at IS NOT NULL` would retain the row forever and fail here. +- A `cancelled` run older than the window is deleted together with its satellite + rows, and a `cancelled` run inside the floor is retained (001.2). Seed it + through the real cancel path, not by writing the status directly, so the test + fails if that path stops stamping the completion timestamp. +- A `runs` row holding a status in neither the history set nor the live-state set + survives every sweep at any age and raises + `office_retention_unknown_status:runs` (001.10). Three companions: the same row + raises the same issue with retention **disabled**, where no sweep runs at all; + a table holding two unrecognized statuses raises **one** issue listing both in + ascending order with their counts; and a test asserts the two status sets + together cover every value in `office/models/enums.go` *and* every status + literal written by SQL in `internal/runs/repository/sqlite`, so a future sixth + value cannot be added silently — the exact gap through which `cancelled` was + missed. +- With more eligible rows than the batch limit, the sweep deletes the oldest + eligible rows first, and the same seed data yields the identical deleted set on + SQLite and PostgreSQL (002.3, 005.1). Without the outer `ORDER BY` this test is + the one that fails. +- A row that becomes floor-protected between selection and delete is not deleted + by the `runs` path (002.4). + +- A batch abandoned after its retry is reported as that table's failure and not + as backlog (002.7). + +- Two backends against one PostgreSQL run one sweep, and the loser records a + skip (002.12). The seed must give the winner **more than one table** to sweep, + so that a transaction-scoped lock — which would release between tables and let + the loser in — fails this test rather than passing it. A companion test drops + the winner's lock connection mid-sweep and asserts it stops before the next + table and does not re-acquire. +- A `runs` batch abandoned after its retry reports zero rows deleted for + `run_events`, `office_run_route_attempts` and `office_run_skills`, not the + counts the rolled-back statements addressed (002.7, 004.6). + +Commands: + +``` +cd apps/backend +go test ./internal/office/... ./internal/runs/... -race -count=1 +go run ./cmd/sqlguard ./internal +go test -race ./internal/persistence/storeconformance -count=1 +KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... \ + ./internal/persistence/storeconformance -count=1 +cd apps/web && pnpm run typecheck && pnpm run i18n:check +``` + +The PostgreSQL line is not optional. The repository's dialect-sensitive suites +self-skip when `KANDEV_TEST_POSTGRES_DSN` is unset, so a green local run without +it proves nothing about AC-OFFICE-RUN-HISTORY-RETENTION-005.1. The engine-parity +test asserts identical deleted sets, retained sets, and counts from identical +seed data on both engines. + +## Rejected alternatives + +- **Wire the existing `CleanExpired` and stop.** One line, and it orphans every + `run_events` row it passes, permanently and invisibly, because no foreign key + on either engine would clean up after it. +- **Add `ON DELETE CASCADE` to the satellite tables instead.** A schema change + to three tables on two engines, requiring a table rebuild on SQLite, to avoid + three `DELETE` statements. It also hides the deletion from the code that has + to count it for the sweep report. +- **Age-prune `run_events` directly.** The obvious implementation, and the one + that breaks the run detail view's incremental tail and can restart a live + run's sequence at zero. Forbidden by 001.6. +- **Run the sweep on the 5s Office tick.** Rejected in 002.1. +- **Count-only retention, keep newest N per owner with no window.** Bounds the + table but makes "how long is my history kept" unanswerable for an install + with mixed routine frequencies. The floor covers the case count-only is good + at; the window covers the case it is bad at. +- **Deletion off by default.** Safe, and it means the gap stays open on every + install that never visits the settings page. The per-table preview, designed + in [run history retention + operations](run-history-retention-operations.md), gives the same protection + without that outcome. +- **Persist sweep history.** Named in the operations requirement's exclusions. + +- **Rely on the window function's `ORDER BY` to order the batch.** It ranks rows + inside each partition; it does not order the rows the outer `LIMIT` draws + from. Leaving the outer select unordered makes a backlog sweep's deleted set + engine-dependent and quietly breaks + AC-OFFICE-RUN-HISTORY-RETENTION-005.1 only on installs large enough to exceed + the batch limit — the installs that need this feature most. + +## Prior art, applied + +**Wiki: unavailable** — receipt in the requirement document. Nothing here should +be read as departing from a wiki position, because none could be consulted. + +**`internal/automation/run_retention.go`** is the closest in-repo precedent, and it +prunes worktrees rather than rows. Three things are taken from it: retention scoped +per owner rather than globally, so one noisy owner cannot evict a quiet one's +only record; a bounded sweep window so a backlog drains across sweeps instead +of walking the whole table each time; and re-checking liveness immediately +before the destructive act, which appears here as the re-asserted predicate and +the batch rollback. One thing is deliberately not taken: it hangs its sweep off +a finalization hook, which ties cleanup frequency to firing frequency and leaves +a stopped routine's history untouched forever. This design uses a clock. + +**`internal/system/storage`** supplies the scheduler shape, the settings +storage and normalization pattern, the hours-based interval with min and max +bounds, and the `health.Checker` route to a production-visible warning. + +**GitLab Duo** and **the Claude apps gateway** are surveyed in the requirement +document's Prior art. What this design takes from them: the 30-day default +window, and the history-vs-live-state split (their per-table windows with one +table marked "until deleted via the API" rather than given a window). Neither +previews before the first deletion; that addition is ours, and it exists because +this ships enabled by default onto installs that already hold history. From 1e7c6e6d05a65b3cd4c3ae6d2ed0a2de4c785429 Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Thu, 10 Sep 2026 08:08:25 +0800 Subject: [PATCH 2/8] fix(office): apply retention window cutoff and report unknown-status counts The sweep passed now as the deletion cutoff instead of now minus the configured window, so the retention window was never actually enforced. Unrecognized status warnings dropped each status's row count, and the retention routes mounted GET alongside the admin-only PUT instead of following the read/admin split every other System-pages surface uses. Adds regression coverage proving the concurrency-gate fingerprint lookup and the task-closure linked-task lookup are unaffected by a sweep, and seeds office_run_skills in the cascade-delete test. --- apps/backend/internal/backendapp/helpers.go | 17 +-- .../internal/office/retention/census.go | 31 +++-- .../internal/office/retention/census_test.go | 6 +- .../internal/office/retention/handler.go | 11 +- .../internal/office/retention/handler_test.go | 39 +++++- .../internal/office/retention/health.go | 13 +- .../internal/office/retention/health_test.go | 21 +++- .../office/retention/store_census_test.go | 14 ++- .../internal/office/retention/store_test.go | 31 ++++- .../internal/office/retention/sweep.go | 22 +++- .../retention/sweep_lookup_integrity_test.go | 113 ++++++++++++++++++ .../internal/office/retention/sweep_test.go | 64 ++++++++++ .../system/retention-settings-card.test.tsx | 23 ++++ .../system/retention-settings-card.tsx | 8 +- .../tests/system/retention-settings.spec.ts | 97 +++++++++++++++ apps/web/lib/types/system.ts | 7 +- 16 files changed, 466 insertions(+), 51 deletions(-) create mode 100644 apps/backend/internal/office/retention/sweep_lookup_integrity_test.go create mode 100644 apps/web/e2e/tests/system/retention-settings.spec.ts diff --git a/apps/backend/internal/backendapp/helpers.go b/apps/backend/internal/backendapp/helpers.go index fb28344d202..d06ea408750 100644 --- a/apps/backend/internal/backendapp/helpers.go +++ b/apps/backend/internal/backendapp/helpers.go @@ -1725,17 +1725,20 @@ func registerSystemRoutes(p routeParams) { } // registerRetentionRoutes mounts GET/PUT /api/v1/system/retention. It is a -// separate admin-scoped group from systemSvc's own /api/v1/system group -// (rather than a field on system.Service) because internal/office/retention -// cannot be imported by internal/system without inverting the existing -// system -> office dependency direction; gin allows two RouterGroups to -// share a path prefix as long as no route collides, and none does here. +// separate group from systemSvc's own /api/v1/system group (rather than a +// field on system.Service) because internal/office/retention cannot be +// imported by internal/system without inverting the existing system -> +// office dependency direction; gin allows two RouterGroups to share a path +// prefix as long as no route collides, and none does here. Read/admin +// split mirrors system.Service.RegisterRoutes: GET is member-readable, +// PUT requires the admin-scoped settings-manage permission. func registerRetentionRoutes(p routeParams) { if p.services == nil || p.services.Retention == nil { return } - admin := p.router.Group("/api/v1/system", authz.RequireOrgScope(authz.ScopeOrgSettingsManage)) - retention.RegisterRoutes(admin, p.services.Retention.Handler) + read := p.router.Group("/api/v1/system") + admin := read.Group("", authz.RequireOrgScope(authz.ScopeOrgSettingsManage)) + retention.RegisterRoutes(read, admin, p.services.Retention.Handler) } // registerHealthRoutes sets up the system health endpoint with all health checkers. diff --git a/apps/backend/internal/office/retention/census.go b/apps/backend/internal/office/retention/census.go index 9a8fa18d575..061f66f2561 100644 --- a/apps/backend/internal/office/retention/census.go +++ b/apps/backend/internal/office/retention/census.go @@ -63,14 +63,22 @@ func (s *CensusState) UnmarshalJSON(data []byte) error { return fmt.Errorf("retention: unknown census state %q", name) } +// UnknownStatusCount is one status this package does not recognize, with +// the number of retained rows currently holding it +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +type UnknownStatusCount struct { + Status string `json:"status"` + Count int64 `json:"count"` +} + // TableCensus is one thresholded table's retained-count evaluation. type TableCensus struct { - State CensusState `json:"state"` - RetainedCount int64 `json:"retained_count"` - AsOf time.Time `json:"as_of"` - UnknownStatuses []string `json:"unknown_statuses,omitempty"` // ascending; only populated for status-bearing tables - TopRoutineID string `json:"top_routine_id,omitempty"` // office_routine_runs only; empty when not applicable - TopRoutineShare float64 `json:"top_routine_share,omitempty"` // top routine's retained rows / table's retained count + State CensusState `json:"state"` + RetainedCount int64 `json:"retained_count"` + AsOf time.Time `json:"as_of"` + UnknownStatuses []UnknownStatusCount `json:"unknown_statuses,omitempty"` // ascending by status; only populated for status-bearing tables + TopRoutineID string `json:"top_routine_id,omitempty"` // office_routine_runs only; empty when not applicable + TopRoutineShare float64 `json:"top_routine_share,omitempty"` // top routine's retained rows / table's retained count } // RetainedCounts holds the current census result for every thresholded @@ -83,9 +91,10 @@ type RetainedCounts struct { // summarizeStatusCensus turns a status->count breakdown into a retained // count (the sum across every status, since "retained" is the table's -// current row count) and the sorted list of statuses belonging to neither -// the history nor the live-state set (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). -func summarizeStatusCensus(counts map[string]int64, history, live []string) (retained int64, unknown []string) { +// current row count) and the status/count list, sorted ascending by +// status, for statuses belonging to neither the history nor the +// live-state set (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +func summarizeStatusCensus(counts map[string]int64, history, live []string) (retained int64, unknown []UnknownStatusCount) { known := make(map[string]bool, len(history)+len(live)) for _, s := range history { known[s] = true @@ -96,10 +105,10 @@ func summarizeStatusCensus(counts map[string]int64, history, live []string) (ret for status, count := range counts { retained += count if !known[status] { - unknown = append(unknown, status) + unknown = append(unknown, UnknownStatusCount{Status: status, Count: count}) } } - sort.Strings(unknown) + sort.Slice(unknown, func(i, j int) bool { return unknown[i].Status < unknown[j].Status }) return retained, unknown } diff --git a/apps/backend/internal/office/retention/census_test.go b/apps/backend/internal/office/retention/census_test.go index 0f82d9b60f1..c776d5bf5e3 100644 --- a/apps/backend/internal/office/retention/census_test.go +++ b/apps/backend/internal/office/retention/census_test.go @@ -32,7 +32,7 @@ func TestSummarizeStatusCensus_DetectsUnknownStatusesSortedAscending(t *testing. if retained != 4 { t.Fatalf("retained = %d, want 4", retained) } - want := []string{"alpha", "zeta"} + want := []UnknownStatusCount{{Status: "alpha", Count: 1}, {Status: "zeta", Count: 1}} if !reflect.DeepEqual(unknown, want) { t.Fatalf("unknown = %v, want %v", unknown, want) } @@ -94,7 +94,7 @@ func TestCensusTracker_FailedEvaluationAfterSuccessKeepsLastCountsMarkedStale(t first := TableCensus{ RetainedCount: 100, AsOf: time.Now().UTC(), - UnknownStatuses: []string{"weird"}, + UnknownStatuses: []UnknownStatusCount{{Status: "weird", Count: 1}}, TopRoutineID: "r-1", TopRoutineShare: 0.5, } @@ -109,7 +109,7 @@ func TestCensusTracker_FailedEvaluationAfterSuccessKeepsLastCountsMarkedStale(t if got.RetainedCount != 100 { t.Fatalf("retained = %d, want 100 (carried over)", got.RetainedCount) } - if !reflect.DeepEqual(got.UnknownStatuses, []string{"weird"}) { + if !reflect.DeepEqual(got.UnknownStatuses, []UnknownStatusCount{{Status: "weird", Count: 1}}) { t.Fatalf("unknownStatuses = %v, want carried over", got.UnknownStatuses) } if got.TopRoutineID != "r-1" || got.TopRoutineShare != 0.5 { diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go index 0b232301fac..13db7a17f14 100644 --- a/apps/backend/internal/office/retention/handler.go +++ b/apps/backend/internal/office/retention/handler.go @@ -37,11 +37,12 @@ func (h *Handler) logError(message string, err error) { } } -// RegisterRoutes wires GET/PUT /api/v1/system/retention. Both routes are -// admin-scoped, and GET is readable while retention is disabled -// (AC-OFFICE-RUN-HISTORY-RETENTION-004.8). -func RegisterRoutes(admin *gin.RouterGroup, handler *Handler) { - admin.GET("/retention", handler.getRetention) +// RegisterRoutes wires GET/PUT /api/v1/system/retention: GET is +// member-readable, matching every sibling System-pages surface (storage, +// queue settings, sleep inhibition), and readable while retention is +// disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8); PUT is admin-scoped. +func RegisterRoutes(read, admin *gin.RouterGroup, handler *Handler) { + read.GET("/retention", handler.getRetention) admin.PUT("/retention", handler.putRetention) } diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go index 8a0fffd8acb..d2a6ebbea3c 100644 --- a/apps/backend/internal/office/retention/handler_test.go +++ b/apps/backend/internal/office/retention/handler_test.go @@ -13,17 +13,22 @@ import ( "github.com/kandev/kandev/internal/auth/authn" ) -// newTestRetentionRouter mirrors production wiring: one admin group guarded -// by authn.RequireAdmin, since both GET and PUT /api/v1/system/retention are -// admin-scoped (unlike storage's split read/admin groups). +// newTestRetentionRouter mirrors production wiring: GET is member-readable, +// PUT requires admin, matching storage's and sleep-inhibition's split +// read/admin groups. func newTestRetentionRouter(handler *Handler) *gin.Engine { + return newTestRetentionRouterAs(handler, authn.RoleAdmin) +} + +func newTestRetentionRouterAs(handler *Handler, role authn.Role) *gin.Engine { router := gin.New() router.Use(func(c *gin.Context) { - authn.SetOnGin(c, authn.Identity{UserID: "admin-1", Role: authn.RoleAdmin}) + authn.SetOnGin(c, authn.Identity{UserID: "user-1", Role: role}) c.Next() }) - admin := router.Group("/api/v1/system", authn.RequireAdmin()) - RegisterRoutes(admin, handler) + read := router.Group("/api/v1/system") + admin := read.Group("", authn.RequireAdmin()) + RegisterRoutes(read, admin, handler) return router } @@ -73,6 +78,28 @@ func TestGetRetention_FreshInstallReturnsDefaultsAndNilLastSweep(t *testing.T) { } } +func TestGetRetention_NonAdminMemberCanRead(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouterAs(handler, authn.RoleMember) + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("member GET status = %d, want 200: %s", response.Code, response.Body.String()) + } +} + +func TestPutRetention_NonAdminMemberIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouterAs(handler, authn.RoleMember) + + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{}`)) + if response.Code != http.StatusForbidden { + t.Fatalf("member PUT status = %d, want 403: %s", response.Code, response.Body.String()) + } +} + func TestGetRetention_ReflectsLastSweepAndCensus(t *testing.T) { gin.SetMode(gin.TestMode) handler, sweeper := newTestHandler(t) diff --git a/apps/backend/internal/office/retention/health.go b/apps/backend/internal/office/retention/health.go index ab8bb4eab9c..01e26a3d465 100644 --- a/apps/backend/internal/office/retention/health.go +++ b/apps/backend/internal/office/retention/health.go @@ -183,7 +183,7 @@ func (c *Checker) censusIssues(settings Settings) []health.Issue { issues = append(issues, issue( fmt.Sprintf("office_retention_unknown_status:%s", e.table), "Unrecognized status in retained rows", - fmt.Sprintf("%s has rows with unrecognized status values, treated as live state and never pruned: %s.", e.table, strings.Join(e.census.UnknownStatuses, ", ")), + fmt.Sprintf("%s has rows with unrecognized status values, treated as live state and never pruned: %s.", e.table, formatUnknownStatuses(e.census.UnknownStatuses)), )) } @@ -208,6 +208,17 @@ func (c *Checker) censusIssues(settings Settings) []health.Issue { return issues } +// formatUnknownStatuses renders each unrecognized status with its row +// count, in the ascending order summarizeStatusCensus already sorted them +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +func formatUnknownStatuses(unknown []UnknownStatusCount) string { + parts := make([]string, len(unknown)) + for i, u := range unknown { + parts[i] = fmt.Sprintf("%s (%d)", u.Status, u.Count) + } + return strings.Join(parts, ", ") +} + func thresholdMessage(table TableName, census TableCensus, warnRows int) string { message := fmt.Sprintf("%s has %d retained rows, over its threshold of %d.", table, census.RetainedCount, warnRows) if census.TopRoutineID != "" { diff --git a/apps/backend/internal/office/retention/health_test.go b/apps/backend/internal/office/retention/health_test.go index 22b38178ad2..a768495a26b 100644 --- a/apps/backend/internal/office/retention/health_test.go +++ b/apps/backend/internal/office/retention/health_test.go @@ -3,6 +3,7 @@ package retention import ( "context" "errors" + "strings" "testing" "github.com/jmoiron/sqlx" @@ -36,6 +37,17 @@ func hasIssue(issues []health.Issue, id string) bool { return false } +func issueMessage(t *testing.T, issues []health.Issue, id string) string { + t.Helper() + for _, i := range issues { + if i.ID == id { + return i.Message + } + } + t.Fatalf("no issue with id %q in %v", id, issueIDs(issues)) + return "" +} + func TestChecker_NameAndCategory(t *testing.T) { checker, _, _ := newTestChecker(t) if checker.Name() != "Office run retention" { @@ -227,12 +239,17 @@ func TestChecker_UnknownStatusRaisesWhileDisabledAndBeforeAnySweep(t *testing.T) seedRoutine(t, conn, "r-1") seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2)) sweeper.RunCensus(ctx) // AC-003.11: census runs independent of the sweep/enabled state issues := checker.Check(ctx) - if !hasIssue(issues, "office_retention_unknown_status:office_routine_runs") { - t.Fatalf("issues = %v, want office_retention_unknown_status:office_routine_runs (001.10, disabled, no sweep ever ran)", issueIDs(issues)) + const id = "office_retention_unknown_status:office_routine_runs" + if !hasIssue(issues, id) { + t.Fatalf("issues = %v, want %s (001.10, disabled, no sweep ever ran)", issueIDs(issues), id) + } + if message := issueMessage(t, issues, id); !strings.Contains(message, "quarantined (2)") { + t.Fatalf("message = %q, want it to name the unrecognized status with its row count", message) } } diff --git a/apps/backend/internal/office/retention/store_census_test.go b/apps/backend/internal/office/retention/store_census_test.go index 2b140d9dc22..34eb2f566b7 100644 --- a/apps/backend/internal/office/retention/store_census_test.go +++ b/apps/backend/internal/office/retention/store_census_test.go @@ -54,13 +54,14 @@ func TestCensusRoutineRuns_DetectsUnknownStatus(t *testing.T) { seedRoutine(t, conn, "r-1") seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2)) census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) if err != nil { t.Fatalf("CensusRoutineRuns: %v", err) } - if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0] != "quarantined" { - t.Fatalf("unknownStatuses = %v, want [quarantined]", census.UnknownStatuses) + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "quarantined" || census.UnknownStatuses[0].Count != 2 { + t.Fatalf("unknownStatuses = %v, want [{quarantined 2}]", census.UnknownStatuses) } } @@ -118,16 +119,17 @@ func TestCensusRuns_RetainedCountIsSumOfEveryStatusNoTopAttribution(t *testing.T seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) seedRun(t, conn, newID(), "agent-1", "queued", nil, daysAgo(0)) seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0)) + seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0)) census, err := store.CensusRuns(ctx, conn, time.Now().UTC()) if err != nil { t.Fatalf("CensusRuns: %v", err) } - if census.RetainedCount != 3 { - t.Fatalf("retainedCount = %d, want 3", census.RetainedCount) + if census.RetainedCount != 4 { + t.Fatalf("retainedCount = %d, want 4", census.RetainedCount) } - if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0] != "mystery" { - t.Fatalf("unknownStatuses = %v, want [mystery]", census.UnknownStatuses) + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "mystery" || census.UnknownStatuses[0].Count != 2 { + t.Fatalf("unknownStatuses = %v, want [{mystery 2}]", census.UnknownStatuses) } if census.TopRoutineID != "" { t.Fatalf("topRoutineID = %q, want empty (runs has no routine attribution)", census.TopRoutineID) diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go index 3c34d44afb3..ad9176bc171 100644 --- a/apps/backend/internal/office/retention/store_test.go +++ b/apps/backend/internal/office/retention/store_test.go @@ -51,6 +51,19 @@ func seedRoutineRun(t *testing.T, conn *sqlx.DB, id, routineID, status string, c } } +func seedRoutineRunWithFingerprintAndLinkedTask( + t *testing.T, conn *sqlx.DB, id, routineID, status string, + completedAt *time.Time, createdAt time.Time, fingerprint, linkedTaskID string, +) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at, dispatch_fingerprint, linked_task_id) + VALUES (?, ?, 'trigger', ?, ?, ?, ?, ?) + `), id, routineID, status, completedAt, createdAt, fingerprint, linkedTaskID); err != nil { + t.Fatalf("seed routine run %s: %v", id, err) + } +} + func seedRun(t *testing.T, conn *sqlx.DB, id, agentProfileID, status string, finishedAt *time.Time, requestedAt time.Time) { t.Helper() if _, err := conn.Exec(conn.Rebind(` @@ -81,6 +94,16 @@ func seedRouteAttempt(t *testing.T, conn *sqlx.DB, runID string, seq int) { } } +func seedRunSkill(t *testing.T, conn *sqlx.DB, runID, skillID string) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_run_skills (run_id, skill_id, version, content_hash, materialized_path) + VALUES (?, ?, 'v1', 'hash', '/path') + `), runID, skillID); err != nil { + t.Fatalf("seed run skill for %s: %v", runID, err) + } +} + func countRows(t *testing.T, conn *sqlx.DB, query string, args ...any) int64 { t.Helper() var n int64 @@ -206,13 +229,14 @@ func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { seedRunEvent(t, conn, runID, 0) seedRunEvent(t, conn, runID, 1) seedRouteAttempt(t, conn, runID, 0) + seedRunSkill(t, conn, runID, "skill-1") result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) if err != nil { t.Fatalf("DeleteRunBatch: %v", err) } - if result.RunsDeleted != 1 || result.RunEventsDeleted != 2 || result.RouteAttemptsDeleted != 1 { - t.Fatalf("result = %+v, want RunsDeleted=1 RunEventsDeleted=2 RouteAttemptsDeleted=1", result) + if result.RunsDeleted != 1 || result.RunEventsDeleted != 2 || result.RouteAttemptsDeleted != 1 || result.RunSkillsDeleted != 1 { + t.Fatalf("result = %+v, want RunsDeleted=1 RunEventsDeleted=2 RouteAttemptsDeleted=1 RunSkillsDeleted=1", result) } if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 0 { t.Fatalf("%d run_events rows remain referencing a deleted run", n) @@ -220,6 +244,9 @@ func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_route_attempts WHERE run_id = ?`, runID); n != 0 { t.Fatalf("%d route attempt rows remain referencing a deleted run", n) } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_skills WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d run skill rows remain referencing a deleted run", n) + } } // TestDeleteRunBatch_SurvivingRunKeepsEveryEvent proves diff --git a/apps/backend/internal/office/retention/sweep.go b/apps/backend/internal/office/retention/sweep.go index 76134a76ffd..8c0c36a6fd2 100644 --- a/apps/backend/internal/office/retention/sweep.go +++ b/apps/backend/internal/office/retention/sweep.go @@ -253,17 +253,18 @@ func (s *Sweeper) isPreviewed(ctx context.Context, q queryer, table TableName) b } func (s *Sweeper) sweepRoutineRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) SweptTableResult { + cutoff := retentionCutoff(now, cfg.WindowDays) if !s.isPreviewed(ctx, q, TableOfficeRoutineRuns) { return s.previewTable(ctx, q, TableOfficeRoutineRuns, func() (int64, error) { - return s.store.CountEligibleRoutineRuns(ctx, q, now, cfg.FloorPerOwner) + return s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner) }, now) } - eligible, err := s.store.CountEligibleRoutineRuns(ctx, q, now, cfg.FloorPerOwner) + eligible, err := s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner) if err != nil { return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} } - deleted, err := s.store.DeleteRoutineRunsBatch(ctx, q, now, cfg.FloorPerOwner, batchLimit) + deleted, err := s.store.DeleteRoutineRunsBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit) if err != nil { return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} } @@ -274,18 +275,19 @@ func (s *Sweeper) sweepRoutineRuns(ctx context.Context, q queryer, cfg TableSett } func (s *Sweeper) sweepRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) (SweptTableResult, satelliteResults) { + cutoff := retentionCutoff(now, cfg.WindowDays) if !s.isPreviewed(ctx, q, TableRuns) { result := s.previewTable(ctx, q, TableRuns, func() (int64, error) { - return s.store.CountEligibleRuns(ctx, q, now, cfg.FloorPerOwner) + return s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner) }, now) return result, satelliteResults{} } - eligible, err := s.store.CountEligibleRuns(ctx, q, now, cfg.FloorPerOwner) + eligible, err := s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner) if err != nil { return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} } - result, err := s.store.DeleteRunBatch(ctx, q, now, cfg.FloorPerOwner, batchLimit) + result, err := s.store.DeleteRunBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit) if err != nil { return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} } @@ -307,6 +309,14 @@ func (s *Sweeper) sweepRuns(ctx context.Context, q queryer, cfg TableSettings, n } } +// retentionCutoff turns a table's configured window into the instant a +// history row's completion time must be older than to be eligible, +// derived from the one sweep-start instant both tables share +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.11). +func retentionCutoff(now time.Time, windowDays int) time.Time { + return now.AddDate(0, 0, -windowDays) +} + // previewTable runs a table's first-ever preview pass: count eligible rows // uncapped, mark the table previewed, and delete nothing // (AC-OFFICE-RUN-HISTORY-RETENTION-003.2, -003.9). diff --git a/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go new file mode 100644 index 00000000000..7c9ed78b9a6 --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go @@ -0,0 +1,113 @@ +package retention + +import ( + "context" + "testing" + + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep proves +// AC-OFFICE-RUN-HISTORY-RETENTION-005.3: after a sweep deletes unrelated +// history rows, including ones sharing the same routine and fingerprint, +// the concurrency gate's fingerprint lookup still returns exactly the live +// task_created run it would have returned had no sweep run. +func TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + repo, err := officesqlite.NewWithDB(conn, conn, nil) + if err != nil { + t.Fatalf("open repo: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "done", &old, old, "fp-1", "") + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "failed", &old, old, "fp-1", "") + + activeID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, activeID, "r-1", "task_created", nil, daysAgo(1), "fp-1", "") + + before, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1") + if err != nil { + t.Fatalf("GetActiveRunForFingerprint (before): %v", err) + } + if before == nil || before.ID != activeID { + t.Fatalf("GetActiveRunForFingerprint (before) = %+v, want id %s", before, activeID) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE status IN ('done', 'failed')`); n != 0 { + t.Fatalf("old rows remaining = %d, want 0 (sweep should have deleted them)", n) + } + + after, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1") + if err != nil { + t.Fatalf("GetActiveRunForFingerprint (after): %v", err) + } + if after == nil || after.ID != activeID { + t.Fatalf("GetActiveRunForFingerprint (after sweep) = %+v, want the same active run %s", after, activeID) + } +} + +// TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing proves +// AC-OFFICE-RUN-HISTORY-RETENTION-005.4: once a sweep deletes the routine +// run linked to a task, the task-closure lookup for that task ID resolves +// to nothing rather than to an unrelated run. +func TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + repo, err := officesqlite.NewWithDB(conn, conn, nil) + if err != nil { + t.Fatalf("open repo: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + deletedID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, deletedID, "r-1", "done", &old, old, "fp-1", "task-1") + + // A second, unrelated run under a different task must not be + // mistaken for task-1's closed-out run. + other := daysAgo(1) + otherID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, otherID, "r-1", "done", &other, other, "fp-2", "task-2") + + before, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (before): %v", err) + } + if before == nil || before.ID != deletedID { + t.Fatalf("GetRoutineRunByLinkedTaskID (before) = %+v, want id %s", before, deletedID) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, deletedID); n != 0 { + t.Fatalf("deleted run rows = %d, want 0", n) + } + + after, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (after): %v", err) + } + if after != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (after sweep) = %+v, want nil", after) + } + + // task-2's own run must be unaffected by task-1's deletion. + stillThere, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-2") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (task-2): %v", err) + } + if stillThere == nil || stillThere.ID != otherID { + t.Fatalf("GetRoutineRunByLinkedTaskID (task-2) = %+v, want id %s", stillThere, otherID) + } +} diff --git a/apps/backend/internal/office/retention/sweep_test.go b/apps/backend/internal/office/retention/sweep_test.go index 1f8de95995a..45f902e0a4a 100644 --- a/apps/backend/internal/office/retention/sweep_test.go +++ b/apps/backend/internal/office/retention/sweep_test.go @@ -120,6 +120,70 @@ func TestRunSweep_SecondPassDeletesAfterPreview(t *testing.T) { } } +// TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow seeds one row +// inside the configured window and one past it, per table, with the floor +// dropped to 0 so only age decides eligibility. A cutoff that ignores +// WindowDays (using the sweep instant itself) would preview and then delete +// both rows instead of only the old one. +func TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) // DefaultSettings() keeps the 30-day window + + seedRoutine(t, conn, "r-1") + recent := daysAgo(1) + old := daysAgo(60) + + recentRoutineRunID := newID() + oldRoutineRunID := newID() + seedRoutineRun(t, conn, recentRoutineRunID, "r-1", "done", &recent, recent) + seedRoutineRun(t, conn, oldRoutineRunID, "r-1", "done", &old, old) + + recentRunID := newID() + oldRunID := newID() + seedRun(t, conn, recentRunID, "agent-1", "finished", &recent, recent) + seedRun(t, conn, oldRunID, "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) // preview pass + + preview, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if preview.OfficeRoutineRuns.WouldDelete != 1 { + t.Fatalf("office_routine_runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.OfficeRoutineRuns.WouldDelete) + } + if preview.Runs.WouldDelete != 1 { + t.Fatalf("runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.Runs.WouldDelete) + } + + sweeper.RunSweep(ctx) // deleting pass + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if last.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.OfficeRoutineRuns.Deleted) + } + if last.Runs.Deleted != 1 { + t.Fatalf("runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.Runs.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, recentRoutineRunID); n != 1 { + t.Fatalf("recent routine-run rows = %d, want 1: a row inside the retention window must survive", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, oldRoutineRunID); n != 0 { + t.Fatalf("old routine-run rows = %d, want 0", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, recentRunID); n != 1 { + t.Fatalf("recent run rows = %d, want 1: a row inside the retention window must survive", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, oldRunID); n != 0 { + t.Fatalf("old run rows = %d, want 0", n) + } +} + func TestRunSweep_BacklogFlaggedWhenEligibleExceedsBatchLimit(t *testing.T) { sweeper, conn := newTestSweeper(t) ctx := context.Background() diff --git a/apps/web/components/settings/system/retention-settings-card.test.tsx b/apps/web/components/settings/system/retention-settings-card.test.tsx index b701f451043..29c61eeb7d9 100644 --- a/apps/web/components/settings/system/retention-settings-card.test.tsx +++ b/apps/web/components/settings/system/retention-settings-card.test.tsx @@ -194,3 +194,26 @@ describe("RetentionSettingsCard", () => { expect(screen.getByTestId("retention-skip-count").textContent).toContain("2"); }); }); + +describe("RetentionSettingsCard unknown-status reporting", () => { + it("renders each unrecognized status with its row count", async () => { + fetchRetentionStatusMock.mockResolvedValue( + statusOf({ + retained_counts: { + office_routine_runs: { + state: "fresh", + retained_count: 5, + as_of: "2026-09-01T00:00:00Z", + unknown_statuses: [{ status: "quarantined", count: 2 }], + }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + renderCard(); + + const row = await screen.findByTestId("retention-retained-office_routine_runs"); + expect(row.textContent).toContain("quarantined (2)"); + }); +}); diff --git a/apps/web/components/settings/system/retention-settings-card.tsx b/apps/web/components/settings/system/retention-settings-card.tsx index 282979a07f2..c08b97fb35f 100644 --- a/apps/web/components/settings/system/retention-settings-card.tsx +++ b/apps/web/components/settings/system/retention-settings-card.tsx @@ -26,12 +26,17 @@ import type { RetentionTableCensus, RetentionTableSweepResult, RetentionSweptTableResult, + RetentionUnknownStatusCount, } from "@/lib/types/system"; function serialize(settings: RetentionSettings | null): string { return settings ? JSON.stringify(settings) : "loading"; } +function formatUnknownStatuses(unknown: RetentionUnknownStatusCount[]): string { + return unknown.map((u) => `${u.status} (${u.count})`).join(", "); +} + function NumberField({ label, help, @@ -452,7 +457,8 @@ function RetainedCountRow({ label, census }: { label: string; census: RetentionT )} {census.unknown_statuses && census.unknown_statuses.length > 0 && ( - {t("system:retentionUnknownStatusesLabel")}: {census.unknown_statuses.join(", ")} + {t("system:retentionUnknownStatusesLabel")}:{" "} + {formatUnknownStatuses(census.unknown_statuses)} )} {census.top_routine_id && ( diff --git a/apps/web/e2e/tests/system/retention-settings.spec.ts b/apps/web/e2e/tests/system/retention-settings.spec.ts new file mode 100644 index 00000000000..08a75c77394 --- /dev/null +++ b/apps/web/e2e/tests/system/retention-settings.spec.ts @@ -0,0 +1,97 @@ +import { test, expect } from "../../fixtures/test-base"; + +/** + * Office run-history retention (docs/specs/office/requirements/run-history-retention*.md). + * The sweep itself is time- and seed-dependent and is covered far more cheaply by backend + * tests; the two honest E2E candidates are the settings round-trip through the real API + * and a threshold warning actually reaching the Health card. Both tests restore whatever + * global retention settings they change, since retention settings are process-global and + * this worker's backend is reused by every test file in the shard. + */ +test.describe("System retention settings", () => { + test("persists an admin policy edit through save and across a backend restart", async ({ + testPage, + backend, + }) => { + test.setTimeout(90_000); + await testPage.goto("/settings/system/data-storage"); + const batchLimit = testPage.getByTestId("retention-batch-limit"); + await expect(batchLimit).toBeVisible(); + const original = await batchLimit.inputValue(); + const updated = original === "7777" ? "8888" : "7777"; + + try { + await batchLimit.fill(updated); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + + await testPage.reload(); + await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated); + + await backend.restart(); + await testPage.reload(); + await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated); + } finally { + await testPage.getByTestId("retention-batch-limit").fill(original); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + } + }); + + test("shows a threshold warning on the Health card once retained runs exceed the configured limit", async ({ + testPage, + backend, + apiClient, + seedData, + }) => { + test.setTimeout(90_000); + await testPage.goto("/settings/system/data-storage"); + const warnField = testPage.getByTestId("retention-runs-warn-rows"); + await expect(warnField).toBeVisible(); + const originalWarnRows = await warnField.inputValue(); + + const initialStatus = await testPage.evaluate(async () => { + const response = await fetch("/api/v1/system/retention"); + return response.json(); + }); + const baselineRetained = + initialStatus.retained_counts.runs.state === "fresh" + ? initialStatus.retained_counts.runs.retained_count + : 0; + // warn_rows=0 disables the threshold check entirely (AC-OFFICE-RUN-HISTORY-RETENTION-004.3), + // and seeding 3 new terminal runs guarantees the post-seed count clears baseline+1 even when + // baseline is 0. + const seededRunCount = 3; + const newWarnRows = baselineRetained + 1; + const expectedRetained = baselineRetained + seededRunCount; + for (let i = 0; i < seededRunCount; i++) { + await apiClient.seedRun({ agentProfileId: seedData.agentProfileId, status: "finished" }); + } + + try { + await warnField.fill(String(newWarnRows)); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + + // The census only re-evaluates on its interval timer or at scheduler Start; a + // restart forces an immediate re-evaluation against the settings just saved and + // the runs just seeded, deterministically, without waiting out the real interval. + await backend.restart(); + + await testPage.goto("/settings/system/status"); + const issue = testPage.getByTestId("system-health-issue-office_retention_threshold:runs"); + await expect(issue).toBeVisible({ timeout: 15_000 }); + await expect(issue).toContainText("over its threshold"); + + await testPage.goto("/settings/system/data-storage"); + const retainedRuns = testPage.getByTestId("retention-retained-runs"); + await expect(retainedRuns).toBeVisible(); + await expect(retainedRuns).toContainText(String(expectedRetained)); + } finally { + await testPage.goto("/settings/system/data-storage"); + await testPage.getByTestId("retention-runs-warn-rows").fill(originalWarnRows); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + } + }); +}); diff --git a/apps/web/lib/types/system.ts b/apps/web/lib/types/system.ts index fad1ad52136..399bc00dd05 100644 --- a/apps/web/lib/types/system.ts +++ b/apps/web/lib/types/system.ts @@ -604,11 +604,16 @@ export interface RetentionLastSweep { export type RetentionCensusState = "not_computed" | "fresh" | "stale"; +export interface RetentionUnknownStatusCount { + status: string; + count: number; +} + export interface RetentionTableCensus { state: RetentionCensusState; retained_count: number; as_of: string; - unknown_statuses?: string[]; + unknown_statuses?: RetentionUnknownStatusCount[]; top_routine_id?: string; top_routine_share?: number; } From fa4163f0a687aac3fcdf79721fcc299cb716a719 Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Thu, 10 Sep 2026 09:04:44 +0800 Subject: [PATCH 3/8] fix(office): start retention scheduler on defaults and fix census race MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Scheduler.Start aborted before spawning its goroutine whenever the stored settings document was unreadable, even though GetSettings already falls back to usable defaults for exactly that case — disabling retention for the rest of the process lifetime with no recovery short of a restart. It now uses whatever Settings GetSettings returns regardless of its error. CensusRoutineRuns computed the retained total and the top-routine attribution from two independent queries, letting a concurrent write between them desynchronize TopRoutineShare from the total it is supposed to be a fraction of. Both numbers now come from one statement. Also renames a store test whose docstring claimed to exercise the delete-time resurrection race but never reached it, and adds coverage for the per-owner/per-agent-profile retention floor with two owners. --- .../internal/office/retention/scheduler.go | 9 +- .../office/retention/scheduler_test.go | 53 ++++++++ .../internal/office/retention/store.go | 59 ++++++--- .../office/retention/store_census_test.go | 74 ++++++++++++ .../internal/office/retention/store_test.go | 113 ++++++++++++++++-- .../office/retention/sweep_postgres_test.go | 37 ++++++ 6 files changed, 318 insertions(+), 27 deletions(-) diff --git a/apps/backend/internal/office/retention/scheduler.go b/apps/backend/internal/office/retention/scheduler.go index 82811e83546..35a035275fe 100644 --- a/apps/backend/internal/office/retention/scheduler.go +++ b/apps/backend/internal/office/retention/scheduler.go @@ -70,10 +70,11 @@ func (s *Scheduler) Start(ctx context.Context) error { return nil } - settings, err := s.settingsStore.GetSettings(ctx) - if err != nil { - return err - } + // GetSettings never fails outright: an unreadable or unparseable stored + // document still yields DefaultSettings, so the scheduler starts on + // those defaults rather than never starting at all. The wrapped error + // is reported separately through Checker.Check. + settings, _ := s.settingsStore.GetSettings(ctx) s.mu.Lock() workerCtx, cancel := context.WithCancel(ctx) diff --git a/apps/backend/internal/office/retention/scheduler_test.go b/apps/backend/internal/office/retention/scheduler_test.go index 0f4406fb63e..2e9fb6ca3cd 100644 --- a/apps/backend/internal/office/retention/scheduler_test.go +++ b/apps/backend/internal/office/retention/scheduler_test.go @@ -5,6 +5,13 @@ import ( "sync" "testing" "time" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" ) // fakeAfter is a deterministic stand-in for time.After, keyed by duration: @@ -248,6 +255,52 @@ func TestScheduler_StartTwiceIsNoop(t *testing.T) { } } +// TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable proves +// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: an unreadable stored settings +// document must not disable retention silently. GetSettings already falls +// back to DefaultSettings on such a document (see settings_store_test.go); +// this proves Start actually uses that fallback and runs the loop instead +// of aborting before the goroutine ever spawns. +func TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable(t *testing.T) { + fake := newFakeAfter() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + if err := settingsRaw.Save(context.Background(), settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + settingsStore := NewSettingsStore(settingsRaw) + sweeper := NewSweeper(pool, NewStore(pool), settingsStore, NewPreviewMarkerStore(settingsRaw)) + scheduler := NewScheduler(settingsStore, sweeper, SchedulerOptions{After: fake.after}) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + // DefaultSettings().Enabled is true, so both timers must arm off the + // fallback defaults rather than the loop never starting at all. + fake.waitArmed(t, sweepInterval(DefaultSettings())) + fake.waitArmed(t, firstSweepDelay) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("office_routine_runs census state = %v, want fresh (Start must run the census off defaults)", counts.OfficeRoutineRuns.State) + } +} + func TestScheduler_StopJoinsWithoutHanging(t *testing.T) { fake := newFakeAfter() settings := DefaultSettings() diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go index eaa1a879d0d..745a9df8fa2 100644 --- a/apps/backend/internal/office/retention/store.go +++ b/apps/backend/internal/office/retention/store.go @@ -3,6 +3,7 @@ package retention import ( "context" "database/sql" + "errors" "fmt" "time" @@ -296,31 +297,46 @@ func (s *Store) CountRunEvents(ctx context.Context, q queryer) (int64, error) { } // CensusRoutineRuns issues office_routine_runs' status census -// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10, -003.5, -003.11): one -// GROUP BY status scan yields both the retained count (the sum across -// every status) and the unknown-status detector, since the sweep's own -// predicate selects only history statuses and therefore cannot observe a -// status outside it. AC-003.5's top-routine attribution needs a -// routine_id dimension the status census does not have, so it is a -// second aggregation rather than reusing the first result set. +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10, -003.5, -003.11). The +// unknown-status detector needs every distinct status value, which the +// top-routine attribution's aggregation does not carry, so it stays a +// separate GROUP BY status scan. The retained total and the top-routine +// attribution, by contrast, must agree with each other by construction — +// AC-003.5's share is defined as one routine's retained rows as a +// proportion of the table's retained count — so routineRunCensusTotals +// reads both from one statement rather than two, which a concurrent write +// between them could otherwise make disagree. func (s *Store) CensusRoutineRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { statusCounts, err := statusCensus(ctx, q, "office_routine_runs") if err != nil { return TableCensus{}, err } - retained, unknown := summarizeStatusCensus(statusCounts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + _, unknown := summarizeStatusCensus(statusCounts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + + if testBetweenRoutineRunCensusReads != nil { + testBetweenRoutineRunCensusReads(q) + } + + retained, topID, topCount, err := routineRunCensusTotals(ctx, q) + if err != nil { + return TableCensus{}, err + } census := TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now} if retained > 0 { - topID, topCount, err := topRoutineByRetainedRows(ctx, q) - if err != nil { - return TableCensus{}, err - } census.TopRoutineID = topID census.TopRoutineShare = float64(topCount) / float64(retained) } return census, nil } +// testBetweenRoutineRunCensusReads, when set, runs right after the +// unknown-status scan and right before routineRunCensusTotals' single- +// statement read — a deterministic seam for proving a concurrent write +// landing there cannot desynchronize the retained total from the +// top-routine attribution, since both now come from that one statement. +// Never set outside tests. +var testBetweenRoutineRunCensusReads func(q queryer) + // CensusRuns issues runs' status census, the runs-table equivalent of // CensusRoutineRuns without the routine attribution AC-003.5 is specific // to office_routine_runs. @@ -362,19 +378,30 @@ func statusCensus(ctx context.Context, q queryer, table string) (map[string]int6 return counts, rows.Err() } -func topRoutineByRetainedRows(ctx context.Context, q queryer) (string, int64, error) { +// routineRunCensusTotals reads office_routine_runs' table-wide retained +// total and the routine holding the largest share of it from one +// statement: a single GROUP BY routine_id pass, with the table total taken +// as a window sum over that same grouping, so the two numbers reflect +// exactly one snapshot and a share computed from them can never exceed 1.0. +// An empty table produces no groups at all; that is the legitimate zero +// state, not a failure, so sql.ErrNoRows is not propagated. +func routineRunCensusTotals(ctx context.Context, q queryer) (retained int64, topRoutineID string, topRoutineCount int64, err error) { var row struct { RoutineID string `db:"routine_id"` Retained int64 `db:"retained"` + Total int64 `db:"total"` } query := ` - SELECT routine_id, COUNT(*) AS retained + SELECT routine_id, COUNT(*) AS retained, SUM(COUNT(*)) OVER () AS total FROM office_routine_runs GROUP BY routine_id ORDER BY retained DESC, routine_id ASC LIMIT 1` if err := q.GetContext(ctx, &row, query); err != nil { - return "", 0, err + if errors.Is(err, sql.ErrNoRows) { + return 0, "", 0, nil + } + return 0, "", 0, err } - return row.RoutineID, row.Retained, nil + return row.Total, row.RoutineID, row.Retained, nil } diff --git a/apps/backend/internal/office/retention/store_census_test.go b/apps/backend/internal/office/retention/store_census_test.go index 34eb2f566b7..a208d79aad8 100644 --- a/apps/backend/internal/office/retention/store_census_test.go +++ b/apps/backend/internal/office/retention/store_census_test.go @@ -158,4 +158,78 @@ func TestCensusRunEvents_PlainCountNoStatusDetection(t *testing.T) { } } +// TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent +// proves the fix for the top-routine attribution's former two-query race: a +// write landing between the unknown-status scan and the single-statement +// totals read must not let the reported TopRoutineShare and RetainedCount +// come from different snapshots of the table. Before the fix, this seam sat +// between two independent reads and could make TopRoutineShare exceed 1.0 +// or attribute a share against a stale total. +func TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutine(t, conn, "r-2") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount) + } + if census.TopRoutineID != "r-2" { + t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 2.0/3.0; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } + if census.TopRoutineShare > 1.0 { + t.Fatalf("topRoutineShare = %v, must never exceed 1.0", census.TopRoutineShare) + } +} + +// TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError +// proves routineRunCensusTotals treats a table that became empty as the +// legitimate zero state rather than propagating sql.ErrNoRows: the old +// two-query design decided whether to run the top-routine query from a +// separately-read, now-stale nonzero total, so this same interleaving used +// to surface an unhandled error instead of a clean zero census. +func TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + if _, err := conn.Exec(`DELETE FROM office_routine_runs`); err != nil { + t.Fatalf("delete all rows: %v", err) + } + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 0 { + t.Fatalf("retainedCount = %d, want 0", census.RetainedCount) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID) + } +} + func timePtr(t time.Time) *time.Time { return &t } diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go index ad9176bc171..fac230eadc7 100644 --- a/apps/backend/internal/office/retention/store_test.go +++ b/apps/backend/internal/office/retention/store_test.go @@ -218,6 +218,58 @@ func TestDeleteRoutineRunsBatch_FloorReassertedAtDeleteTime(t *testing.T) { } } +// TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine proves +// AC-OFFICE-RUN-HISTORY-RETENTION-001.4's floor is per-owner: every seed +// helper elsewhere in this suite uses a single routine, so a regression +// that dropped routineRunEligibleSubquery's PARTITION BY routine_id +// (turning a per-owner floor into one shared across every routine) would +// otherwise leave the whole suite green. +func TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + routineA, routineB := newID(), newID() + seedRoutine(t, conn, routineA) + seedRoutine(t, conn, routineB) + + cutoff := daysAgo(30) + var newestA, newestB string + for i := 0; i < 5; i++ { + completed := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86) + idA, idB := newID(), newID() + seedRoutineRun(t, conn, idA, routineA, "done", &completed, completed) + seedRoutineRun(t, conn, idB, routineB, "done", &completed, completed) + if i == 4 { + newestA, newestB = idA, idB + } + } + + // Floor 3 per routine: each routine has 5 history rows, so 2 are + // eligible per routine, 4 total. A floor shared across both routines + // (10 rows, floor 3) would instead delete 7 and could delete either + // routine's newest row. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 3, 100) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 4 { + t.Fatalf("deleted = %d, want 4 (2 eligible per routine, floor 3 held independently)", deleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineA); n != 3 { + t.Fatalf("routineA remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineB); n != 3 { + t.Fatalf("routineB remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestA); n != 1 { + t.Fatalf("routineA's newest row was deleted; its floor should have protected it") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestB); n != 1 { + t.Fatalf("routineB's newest row was deleted; its floor should have protected it") + } +} + func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { conn := testDB(t) store := NewStore(db.NewPool(conn, conn)) @@ -277,13 +329,15 @@ func TestDeleteRunBatch_SurvivingRunKeepsEveryEvent(t *testing.T) { } } -// TestDeleteRunBatch_ResurrectedRunSurvivesAndKeepsSatellites drives the -// AC-OFFICE-RUN-HISTORY-RETENTION-002.4 race directly: a run selected as -// eligible is resurrected to "queued" (as ScheduleRetry does, clearing -// finished_at) after selection but before the delete statement runs. The -// re-assertion inside DeleteRunBatch's delete step must catch this, -// rolling back and leaving the run and its satellites intact. -func TestDeleteRunBatch_ResurrectedRunSurvivesAndKeepsSatellites(t *testing.T) { +// TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected proves a +// run resurrected to "queued" (as ScheduleRetry does, clearing finished_at) +// before DeleteRunBatch runs at all is excluded by the eligibility +// selection itself, leaving it and its satellites untouched. This is the +// simple case; the delete-time re-assertion this package's AC-002.4 +// re-assertion actually catches — a resurrection landing between +// selection and the delete statement, inside one attempt — is covered by +// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives below. +func TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected(t *testing.T) { conn := testDB(t) store := NewStore(db.NewPool(conn, conn)) ctx := context.Background() @@ -435,6 +489,51 @@ func TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches(t *testing.T) { } } +// TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile is the runs-table +// equivalent of TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine: +// every other DeleteRunBatch test in this file uses a single +// "agent-1" owner, so a regression that dropped runEligibleSubquery's +// PARTITION BY agent_profile_id would otherwise leave the whole suite green. +func TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + cutoff := daysAgo(30) + var newestA, newestB string + for i := 0; i < 5; i++ { + finished := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86) + idA, idB := newID(), newID() + seedRun(t, conn, idA, "agent-a", "finished", &finished, finished) + seedRun(t, conn, idB, "agent-b", "finished", &finished, finished) + if i == 4 { + newestA, newestB = idA, idB + } + } + + // Floor 3 per agent profile: each owns 5 history rows, so 2 are + // eligible per owner, 4 total. + result, err := store.DeleteRunBatch(ctx, conn, cutoff, 3, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 4 || result.Abandoned { + t.Fatalf("result = %+v, want 4 deleted (2 eligible per agent profile, floor 3 held independently)", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-a"); n != 3 { + t.Fatalf("agent-a remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-b"); n != 3 { + t.Fatalf("agent-b remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestA); n != 1 { + t.Fatalf("agent-a's newest run was deleted; its floor should have protected it") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestB); n != 1 { + t.Fatalf("agent-b's newest run was deleted; its floor should have protected it") + } +} + func TestCountRunEvents_PlainCount(t *testing.T) { conn := testDB(t) store := NewStore(db.NewPool(conn, conn)) diff --git a/apps/backend/internal/office/retention/sweep_postgres_test.go b/apps/backend/internal/office/retention/sweep_postgres_test.go index 06bfc9e70bb..06d2d62a7ff 100644 --- a/apps/backend/internal/office/retention/sweep_postgres_test.go +++ b/apps/backend/internal/office/retention/sweep_postgres_test.go @@ -199,3 +199,40 @@ func TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire(t *te } replacement.release() } + +// TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite +// proves routineRunCensusTotals' window-function query — SUM(COUNT(*)) OVER +// () over a GROUP BY routine_id — is valid PostgreSQL and, being one +// statement, cannot be split by a write landing between the unknown-status +// scan and the totals read, unlike the two independent queries it replaced. +func TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + seedRoutine(t, conn, "r-1") + seedRoutine(t, conn, "r-2") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount) + } + if census.TopRoutineID != "r-2" { + t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 2.0/3.0; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } +} From 7a71ce582993b1706ebb75da6a5dab23138b3c43 Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Thu, 10 Sep 2026 09:59:47 +0800 Subject: [PATCH 4/8] fix(office): chunk retention batch deletes and close PUT/delete races Review round 3's mandatory cross-vendor pass found two blocker-level production defects and one major one in the retention sweep, each independently verified by direct code inspection and a targeted regression test before being fixed: - deleteRunBatchOnce bound its whole selected id list as a single IN clause across four statements; batch_limit's documented range (100-100,000) can overflow SQLite's or PostgreSQL's bind-parameter limit. Both the satellite deletes and the runs delete now chunk ids via the same bound this repo already uses elsewhere for the same reason. - The runs DELETE lacked a direct outer status/age predicate, unlike its office_routine_runs sibling. On PostgreSQL, EvalPlanQual's recheck of a concurrently resurrected row re-evaluates direct column predicates but not an uncorrelated id-membership subquery, so a row resurrected between selection and delete could still be deleted. Fixed by mirroring the sibling query's direct predicates; proven with a real two-connection PostgreSQL test that reproduces the race. - Concurrent PUT /retention requests could interleave their save and scheduler-apply steps, leaving the scheduler applying stale settings after a newer write already committed. The handler now serializes each PUT's save+apply as one critical section. Also adds the two test-rigor gaps two independent review legs converged on: SQLite/PostgreSQL backlog-path parity, and per-table independence of preview-marker state when a sibling table's preview fails mid-sweep. Co-Authored-By: Claude Sonnet 5 --- .../internal/office/retention/handler.go | 21 ++ .../internal/office/retention/handler_test.go | 80 ++++++++ .../internal/office/retention/store.go | 111 ++++++++--- .../office/retention/store_postgres_test.go | 184 ++++++++++++++++++ .../internal/office/retention/store_test.go | 50 +++++ .../internal/office/retention/sweep_test.go | 64 ++++++ 6 files changed, 485 insertions(+), 25 deletions(-) create mode 100644 apps/backend/internal/office/retention/store_postgres_test.go diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go index 13db7a17f14..58fd21be233 100644 --- a/apps/backend/internal/office/retention/handler.go +++ b/apps/backend/internal/office/retention/handler.go @@ -3,6 +3,7 @@ package retention import ( "errors" "net/http" + "sync" "time" "github.com/gin-gonic/gin" @@ -24,6 +25,12 @@ type HandlerConfig struct { // Handler serves GET/PUT /api/v1/system/retention. type Handler struct { config HandlerConfig + + // mu serializes a PUT's save and scheduler-apply as one critical + // section, so two concurrent PUTs cannot interleave into the scheduler + // applying the older of the two writes after the newer one is already + // stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.5's last-writer-wins). + mu sync.Mutex } // NewHandler wires a Handler to its dependencies. @@ -91,6 +98,9 @@ func (h *Handler) putRetention(c *gin.Context) { return } + h.mu.Lock() + defer h.mu.Unlock() + saved, err := h.config.SettingsStore.SaveSettings(c.Request.Context(), settings) if err != nil { if errors.Is(err, ErrValidation) { @@ -102,8 +112,19 @@ func (h *Handler) putRetention(c *gin.Context) { return } + if testBetweenSaveAndApply != nil { + testBetweenSaveAndApply() + } + if h.config.OnSettingsChanged != nil { h.config.OnSettingsChanged(saved) } c.JSON(http.StatusOK, saved) } + +// testBetweenSaveAndApply, when set, runs after a PUT's SaveSettings +// commits and before OnSettingsChanged is invoked, while mu is still held — +// a deterministic seam for proving a second PUT cannot save and apply in +// between (the concurrent-PUT desync this mutex exists to prevent). Never +// set outside tests. +var testBetweenSaveAndApply func() diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go index d2a6ebbea3c..8c5611fc731 100644 --- a/apps/backend/internal/office/retention/handler_test.go +++ b/apps/backend/internal/office/retention/handler_test.go @@ -6,7 +6,9 @@ import ( "net/http" "net/http/httptest" "strings" + "sync" "testing" + "time" "github.com/gin-gonic/gin" @@ -310,3 +312,81 @@ func TestPutRetention_SuccessInvokesOnSettingsChanged(t *testing.T) { t.Fatal("OnSettingsChanged received Enabled=true, want false") } } + +// TestPutRetention_ConcurrentPUTsApplyInSaveOrder is the regression test for +// the concurrent-PUT scheduler desync found in review: without serializing +// SaveSettings and OnSettingsChanged as one critical section, a second PUT +// racing between the first's save and apply could complete its own save and +// apply entirely in between, leaving the scheduler applying the first PUT's +// now-stale settings after the second PUT's newer write already committed +// (AC-004.5's last-writer-wins). testBetweenSaveAndApply fires while the +// first PUT still holds the handler's mutex; it starts a second PUT +// concurrently and proves that second PUT cannot complete until the first +// releases the mutex, so the two applications can never interleave. +func TestPutRetention_ConcurrentPUTsApplyInSaveOrder(t *testing.T) { + gin.SetMode(gin.TestMode) + sweeper, _ := newTestSweeper(t) + + var mu sync.Mutex + var applied []bool + handler := NewHandler(HandlerConfig{ + SettingsStore: sweeper.settingsStore, + Sweeper: sweeper, + OnSettingsChanged: func(s Settings) { + mu.Lock() + applied = append(applied, s.Enabled) + mu.Unlock() + }, + }) + router := newTestRetentionRouter(handler) + + secondDone := make(chan struct{}) + secondStarted := false + + t.Cleanup(func() { testBetweenSaveAndApply = nil }) + testBetweenSaveAndApply = func() { + testBetweenSaveAndApply = nil // only race a second request once + secondStarted = true + go func() { + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": true}`)) + if response.Code != http.StatusOK { + t.Errorf("second PUT status = %d, want 200: %s", response.Code, response.Body.String()) + } + close(secondDone) + }() + + select { + case <-secondDone: + t.Fatal("second PUT completed while the first still held the critical section") + case <-time.After(100 * time.Millisecond): + } + } + + first := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": false}`)) + if first.Code != http.StatusOK { + t.Fatalf("first PUT status = %d, want 200: %s", first.Code, first.Body.String()) + } + if !secondStarted { + t.Fatal("test hook never fired; the race was not exercised") + } + + select { + case <-secondDone: + case <-time.After(5 * time.Second): + t.Fatal("second PUT never completed after the first released the critical section") + } + + mu.Lock() + defer mu.Unlock() + if len(applied) != 2 || applied[0] != false || applied[1] != true { + t.Fatalf("OnSettingsChanged calls = %+v, want [false, true] in save order", applied) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if !stored.Enabled { + t.Fatal("stored Enabled = false, want true (the second, later PUT must win)") + } +} diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go index 745a9df8fa2..bcf3fe980b0 100644 --- a/apps/backend/internal/office/retention/store.go +++ b/apps/backend/internal/office/retention/store.go @@ -238,22 +238,7 @@ func (s *Store) deleteRunBatchOnce(ctx context.Context, q queryer, ids []string, return RunBatchResult{}, false, err } - query := ` - DELETE FROM runs - WHERE id IN (?) - AND id IN ( - SELECT id FROM (` + runEligibleSubquery() + `) ranked - WHERE rn > ? AND completion_time < ? - )` - bound, args, err := db.Bind(tx, query, ids, RunHistoryStatuses, floor, cutoff) - if err != nil { - return RunBatchResult{}, false, err - } - res, err := tx.ExecContext(ctx, bound, args...) - if err != nil { - return RunBatchResult{}, false, err - } - runsDeleted, err := res.RowsAffected() + runsDeleted, err := deleteRunsByIDs(ctx, tx, ids, cutoff, floor) if err != nil { return RunBatchResult{}, false, err } @@ -273,17 +258,93 @@ func (s *Store) deleteRunBatchOnce(ctx context.Context, q queryer, ids []string, }, true, nil } -func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) { - query := fmt.Sprintf(`DELETE FROM %s WHERE run_id IN (?)`, table) - bound, args, err := db.Bind(tx, query, ids) - if err != nil { - return 0, err +// retentionMaxHostParams caps id-list placeholders per statement. +// batch_limit's documented range (AC-OFFICE-RUN-HISTORY-RETENTION-004.3) +// permits up to 100,000, which would otherwise bind that many ids in one IN +// clause and can overflow SQLite's compiled variable-count limit or +// PostgreSQL's wire-protocol parameter cap. Matches the bound this repo +// already uses for the same reason (internal/task/repository/sqlite's +// sqliteMaxHostParams). +const retentionMaxHostParams = 500 + +// chunkIDs splits ids into sub-slices of at most size entries so an +// IN-clause query built from them stays under retentionMaxHostParams +// regardless of batch_limit. An empty input returns nil rather than one +// empty chunk, since an empty IN () clause is a SQL syntax error. +func chunkIDs(ids []string, size int) [][]string { + if len(ids) == 0 { + return nil } - res, err := tx.ExecContext(ctx, bound, args...) - if err != nil { - return 0, err + if size <= 0 || len(ids) <= size { + return [][]string{ids} } - return res.RowsAffected() + chunks := make([][]string, 0, (len(ids)+size-1)/size) + for i := 0; i < len(ids); i += size { + end := i + size + if end > len(ids) { + end = len(ids) + } + chunks = append(chunks, ids[i:end]) + } + return chunks +} + +// deleteRunsByIDs deletes ids from runs, in chunks of at most +// retentionMaxHostParams. Each chunk's outer WHERE re-asserts status and age +// directly, in addition to the floor-checking subquery: PostgreSQL's +// EvalPlanQual recheck of a concurrently updated row re-evaluates a direct +// column predicate against the row's fresh values, but does not rebuild an +// uncorrelated id-membership subquery, so the subquery alone is not enough +// to exclude a row resurrected between selection and delete +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4). +func deleteRunsByIDs(ctx context.Context, tx *sqlx.Tx, ids []string, cutoff time.Time, floor int) (int64, error) { + var total int64 + for _, chunk := range chunkIDs(ids, retentionMaxHostParams) { + query := ` + DELETE FROM runs + WHERE id IN (?) + AND id IN ( + SELECT id FROM (` + runEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ) + AND status IN (?) + AND COALESCE(finished_at, requested_at) < ?` + bound, args, err := db.Bind(tx, query, chunk, RunHistoryStatuses, floor, cutoff, RunHistoryStatuses, cutoff) + if err != nil { + return 0, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + affected, err := res.RowsAffected() + if err != nil { + return 0, err + } + total += affected + } + return total, nil +} + +func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) { + var total int64 + for _, chunk := range chunkIDs(ids, retentionMaxHostParams) { + query := fmt.Sprintf(`DELETE FROM %s WHERE run_id IN (?)`, table) + bound, args, err := db.Bind(tx, query, chunk) + if err != nil { + return 0, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + affected, err := res.RowsAffected() + if err != nil { + return 0, err + } + total += affected + } + return total, nil } // CountRunEvents is run_events' plain retained count: it has no status diff --git a/apps/backend/internal/office/retention/store_postgres_test.go b/apps/backend/internal/office/retention/store_postgres_test.go new file mode 100644 index 00000000000..91e870e5786 --- /dev/null +++ b/apps/backend/internal/office/retention/store_postgres_test.go @@ -0,0 +1,184 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/testutil" +) + +// openSharedSchemaPostgresConn opens a second, independent PostgreSQL +// connection pointed at the same isolated test schema as an existing +// connection. A real second physical connection is required to hold a row +// lock that a concurrent statement on the first connection genuinely blocks +// on — a sequential in-process test hook cannot reproduce that. +func openSharedSchemaPostgresConn(t *testing.T, dsn, schema string) *sqlx.DB { + t.Helper() + raw, err := db.OpenPostgres(dsn, 1, 1) + if err != nil { + t.Fatalf("open second postgres connection: %v", err) + } + conn := sqlx.NewDb(raw, "pgx") + conn.SetMaxOpenConns(1) + conn.SetMaxIdleConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := conn.Exec("SET search_path TO " + schema); err != nil { + t.Fatalf("set search_path on second connection: %v", err) + } + return conn +} + +// TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow +// is the regression test for deleteRunsByIDs' missing direct status/age +// predicate (AC-OFFICE-RUN-HISTORY-RETENTION-002.4): a run resurrected by a +// concurrent transaction that has not yet committed when DeleteRunBatch +// selects it, but commits while the DELETE statement is blocked acquiring +// the row's lock — driving PostgreSQL's real EvalPlanQual recheck path. The +// package's existing resurrection tests +// (TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives) use a +// sequential in-process hook that resurrects strictly before the DELETE +// statement starts; they cannot reach this mid-statement window, which +// needs a second, genuinely concurrent connection. +func TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + var schema string + if err := conn.Get(&schema, `SELECT current_schema()`); err != nil { + t.Fatalf("select current_schema: %v", err) + } + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + holder := openSharedSchemaPostgresConn(t, dsn, schema) + holderTx, err := holder.BeginTx(ctx, nil) + if err != nil { + t.Fatalf("begin holder tx: %v", err) + } + // Resurrecting the row inside an uncommitted transaction takes its row + // lock immediately, but the new values are not visible to conn's own + // snapshot until Commit below: selectEligibleRunIDs' plain read is + // never blocked by an uncommitted writer, so it still sees the row as + // terminal and eligible, exactly the window this test targets. + if _, err := holderTx.ExecContext(ctx, `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = $1`, runID); err != nil { + t.Fatalf("resurrect run inside holder tx: %v", err) + } + + admin := testutil.OpenIsolatedPostgres(t, dsn) // separate schema; pg_stat_activity is instance-wide, not schema-scoped + deleteDone := make(chan RunBatchResult, 1) + deleteErr := make(chan error, 1) + go func() { + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + deleteErr <- err + return + } + deleteDone <- result + }() + + // Wait for DeleteRunBatch's DELETE statement to actually be blocked on + // the holder's row lock before committing, so the recheck this test + // targets is guaranteed to happen rather than racing ahead of it. + deadline := time.Now().Add(10 * time.Second) + for { + var waiting bool + if err := admin.GetContext(ctx, &waiting, ` + SELECT EXISTS( + SELECT 1 FROM pg_stat_activity + WHERE wait_event_type = 'Lock' AND query ILIKE '%DELETE FROM runs%' + )`); err != nil { + t.Fatalf("poll pg_stat_activity: %v", err) + } + if waiting { + break + } + if time.Now().After(deadline) { + t.Fatal("DeleteRunBatch's DELETE never showed up waiting on the holder's row lock") + } + time.Sleep(20 * time.Millisecond) + } + + if err := holderTx.Commit(); err != nil { + t.Fatalf("commit holder tx: %v", err) + } + + var result RunBatchResult + select { + case result = <-deleteDone: + case err := <-deleteErr: + t.Fatalf("DeleteRunBatch: %v", err) + case <-time.After(10 * time.Second): + t.Fatal("DeleteRunBatch did not complete after the holder committed") + } + + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op: the row was resurrected before the delete's row lock was granted, so PostgreSQL's own recheck of the row must see it as no longer eligible", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted despite the concurrent recheck") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted despite the concurrent recheck") + } +} + +// TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite is +// AC-OFFICE-RUN-HISTORY-RETENTION-005.1's mandated backlog-path parity test: +// batch selection order is fixed by named columns +// (completion_time ASC, id ASC) rather than left to the engine, so this +// must select and delete the exact same rows on PostgreSQL as +// TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns proves on +// SQLite for the identical settings and starting rows. +func TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + oldest, middle, newest := newID(), newID(), newID() + oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45) + seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC) + seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC) + seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC) + + // floor 0 so all three are eligible; batch limit 2 -> the two oldest + // go, the newest survives as backlog — same as the SQLite test. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 2 { + t.Fatalf("deleted = %d, want 2", deleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest); n != 1 { + t.Fatal("newest row was deleted on PostgreSQL; oldest-first ordering violated") + } + for _, id := range []string{oldest, middle} { + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 { + t.Fatalf("row %s (older) still present on PostgreSQL after batch limit 2", id) + } + } + + eligible, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if eligible != 1 { + t.Fatalf("remaining eligible = %d, want 1 (backlog: the newest row is still eligible, just not yet batched)", eligible) + } +} diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go index fac230eadc7..bb38bbfae10 100644 --- a/apps/backend/internal/office/retention/store_test.go +++ b/apps/backend/internal/office/retention/store_test.go @@ -534,6 +534,56 @@ func TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile(t *testing.T) { } } +// TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements proves +// deleteRunBatchOnce's satellite and runs deletes split their id list into +// chunks of at most retentionMaxHostParams rather than binding the whole +// batch as one IN clause, which is what let batch_limit's documented range +// (AC-OFFICE-RUN-HISTORY-RETENTION-004.3, up to 100,000) overflow a single +// statement's bind-parameter limit on either engine. retentionMaxHostParams+2 +// runs, each with one satellite row apiece, forces the delete loop to span +// more than one chunk; every row and every satellite must still be deleted +// and the reported count must reflect the true total, not just one chunk's. +func TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + const rowCount = retentionMaxHostParams + 2 + finished := daysAgo(60) + ids := make([]string, 0, rowCount) + for i := 0; i < rowCount; i++ { + id := newID() + seedRun(t, conn, id, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, id, 0) + ids = append(ids, id) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, rowCount) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.Abandoned { + t.Fatalf("result = %+v, want a clean delete, not abandoned", result) + } + if result.RunsDeleted != int64(rowCount) { + t.Fatalf("RunsDeleted = %d, want %d", result.RunsDeleted, rowCount) + } + if result.RunEventsDeleted != int64(rowCount) { + t.Fatalf("RunEventsDeleted = %d, want %d", result.RunEventsDeleted, rowCount) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 { + t.Fatalf("runs remaining = %d, want 0 (every row across every chunk must be deleted)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events`); n != 0 { + t.Fatalf("run_events remaining = %d, want 0", n) + } + for _, id := range ids { + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, id); n != 0 { + t.Fatalf("run %s remains after a chunked delete", id) + } + } +} + func TestCountRunEvents_PlainCount(t *testing.T) { conn := testDB(t) store := NewStore(db.NewPool(conn, conn)) diff --git a/apps/backend/internal/office/retention/sweep_test.go b/apps/backend/internal/office/retention/sweep_test.go index 45f902e0a4a..2f4005f11a8 100644 --- a/apps/backend/internal/office/retention/sweep_test.go +++ b/apps/backend/internal/office/retention/sweep_test.go @@ -344,6 +344,70 @@ func TestRunSweep_AbandonedRunsBatchReportsFailureNotBacklogWithZeroSatellites(t } } +// TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState +// is AC-OFFICE-RUN-HISTORY-RETENTION-003.4's per-table independence test: +// office_routine_runs' preview completes successfully, but runs' own preview +// fails in the same sweep (its table is temporarily unreachable). The next +// sweep must not preview office_routine_runs a second time — its preview +// already completed and a sibling's failure must not reopen it — and must +// preview runs again, since its own preview never recorded completion. +func TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + // Hide runs after office_routine_runs' own preview work has already + // completed for this sweep, so runs' preview fails on a genuine SQL + // error rather than a simulated one. + testBetweenTablesSweep = func(queryer) { + conn.MustExec(`ALTER TABLE runs RENAME TO runs_hidden`) + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + sweeper.RunSweep(ctx) + + first, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if !first.OfficeRoutineRuns.Previewed || first.OfficeRoutineRuns.Err != "" { + t.Fatalf("office_routine_runs = %+v, want a clean, completed preview", first.OfficeRoutineRuns) + } + if first.Runs.Err == "" { + t.Fatal("runs.Err is empty, want the preview failure recorded") + } + if first.Runs.Previewed { + t.Fatal("runs.Previewed = true, want false: the preview did not complete") + } + + testBetweenTablesSweep = nil + conn.MustExec(`ALTER TABLE runs_hidden RENAME TO runs`) + + sweeper.RunSweep(ctx) + + second, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if second.OfficeRoutineRuns.Previewed { + t.Fatal("office_routine_runs.Previewed = true on the second sweep, want false: its preview already completed and must not run a second time because a sibling table failed") + } + if second.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1 (it should now be deleting, having already completed its preview)", second.OfficeRoutineRuns.Deleted) + } + if !second.Runs.Previewed || second.Runs.Err != "" { + t.Fatalf("runs = %+v, want a fresh, successful preview: its earlier failed preview must not count as completed", second.Runs) + } + if second.Runs.WouldDelete != 1 { + t.Fatalf("runs.WouldDelete = %d, want 1", second.Runs.WouldDelete) + } +} + func TestLastSweepSnapshot_FalseBeforeFirstSweep(t *testing.T) { sweeper, _ := newTestSweeper(t) if _, ok := sweeper.LastSweepSnapshot(); ok { From 9622d69d4347bcf03ee21478410499b5b9191dc7 Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Thu, 10 Sep 2026 11:56:29 +0800 Subject: [PATCH 5/8] fix(office): fix stale retention draft after save and address PR review findings Fixes a real regression from this branch: the Data & Logs composition test crashed on RetentionSettingsCard's new useIsAdmin() call because its store mock lacked auth state. Also fixes a save/reload race where a failed post-save GET left the store holding pre-save settings while the shared save coordinator believed the save had already succeeded, plus rejects trailing JSON on PUT, adds the package's goleak TestMain, and removes review-round narration from production comments per repo convention. Co-Authored-By: Claude Sonnet 5 --- .../internal/office/retention/goleak_test.go | 15 +++++ .../internal/office/retention/handler_test.go | 20 ++++++ .../internal/office/retention/health.go | 12 ++-- .../backend/internal/office/retention/lock.go | 32 +++++---- .../office/retention/preview_marker.go | 10 +-- .../office/retention/settings_wire.go | 3 + .../internal/office/retention/store.go | 3 + .../internal/office/retention/sweep.go | 21 +++--- .../system/retention-settings-card.test.tsx | 65 ++++++++++++------- .../settings/system/system-route-copy.test.ts | 2 + .../domains/system/use-retention-settings.ts | 24 ++++++- 11 files changed, 143 insertions(+), 64 deletions(-) create mode 100644 apps/backend/internal/office/retention/goleak_test.go diff --git a/apps/backend/internal/office/retention/goleak_test.go b/apps/backend/internal/office/retention/goleak_test.go new file mode 100644 index 00000000000..7cfa26b30a1 --- /dev/null +++ b/apps/backend/internal/office/retention/goleak_test.go @@ -0,0 +1,15 @@ +package retention + +import ( + "testing" + + "go.uber.org/goleak" +) + +// TestMain enforces no goroutine leaks across the retention package. +// Scheduler.Start spawns a single lifecycle-managed loop goroutine; Stop +// cancels its context and waits on the WaitGroup. Regressions where Stop +// forgets to cancel or a test leaves a scheduler running surface here. +func TestMain(m *testing.M) { + goleak.VerifyTestMain(m) +} diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go index 8c5611fc731..29111c353b8 100644 --- a/apps/backend/internal/office/retention/handler_test.go +++ b/apps/backend/internal/office/retention/handler_test.go @@ -231,6 +231,26 @@ func TestPutRetention_UnknownFieldIsRejectedNamingTheField(t *testing.T) { } } +func TestPutRetention_TrailingDataAfterObjectIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": true} {"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults (nothing written on rejection)", stored) + } +} + func TestPutRetention_FractionalNumberIsRejected(t *testing.T) { gin.SetMode(gin.TestMode) handler, _ := newTestHandler(t) diff --git a/apps/backend/internal/office/retention/health.go b/apps/backend/internal/office/retention/health.go index 01e26a3d465..67fd613f5d4 100644 --- a/apps/backend/internal/office/retention/health.go +++ b/apps/backend/internal/office/retention/health.go @@ -22,11 +22,11 @@ const ( // there is no separate stored issue map to keep in sync with them. // // office_retention_count_failed:
is a ninth issue id beyond the -// design's closed eight-id catalogue: F33 (a Build-accepted spec gap) found -// no id for a failed census evaluation. It fires only on CensusStale (a -// table that had a successful evaluation and then failed); CensusNotComputed -// is the pre-first-success state AC-003.11 requires rendering as absent -// rather than alarming, so it raises nothing on its own. +// design's closed eight-id catalogue, covering a failed census evaluation. +// It fires only on CensusStale (a table that had a successful evaluation and +// then failed); CensusNotComputed is the pre-first-success state AC-003.11 +// requires rendering as absent rather than alarming, so it raises nothing on +// its own. // // office_retention_threshold:
and office_retention_disabled:
// both answer AC-003.5/-003.7's "retained count over threshold" condition, @@ -152,7 +152,7 @@ func failedTableIssues(last LastSweep) []health.Issue { } // censusIssues covers AC-001.10 (unknown status), AC-003.5/-003.7 (threshold, -// split on enabled/disabled), and F33's office_retention_count_failed. +// split on enabled/disabled), and office_retention_count_failed. func (c *Checker) censusIssues(settings Settings) []health.Issue { counts := c.sweeper.CensusSnapshot() diff --git a/apps/backend/internal/office/retention/lock.go b/apps/backend/internal/office/retention/lock.go index ab0febeb18e..a96802c3e56 100644 --- a/apps/backend/internal/office/retention/lock.go +++ b/apps/backend/internal/office/retention/lock.go @@ -20,28 +20,25 @@ const advisoryLockKey = "office_run_retention_sweep" const unlockTimeout = 5 * time.Second // sweepSession is a PostgreSQL session-scoped, non-blocking exclusivity -// lock for one sweep (AC-OFFICE-RUN-HISTORY-RETENTION-002.12). Its shape -// resolves three findings the operator accepted as Build's to decide -// (F25, F26, F27 in the task plan): +// lock for one sweep (AC-OFFICE-RUN-HISTORY-RETENTION-002.12): // -// - F27 (single connection budget): every statement the sweep issues — -// count, delete, census — runs through this one dedicated connection, +// - Single connection budget: every statement the sweep issues — count, +// delete, census — runs through this one dedicated connection, // returned by queryer(), rather than reserving it for the lock alone // and running batches on the shared pool. Reserving a second // connection is exactly what a maxOpenConns=1 pool (what // testutil.OpenIsolatedPostgres, the mandated Postgres-gated test // harness, sets) cannot supply; one connection total removes the // deadlock. -// - F25 (exclusivity window): because every sweep statement runs on the -// lock connection, there is no window where work proceeds on a -// different connection after the session died — if the session ends, -// the very next statement on it fails immediately instead of -// continuing to run against the pool while another backend has -// already re-acquired the lock. The between-tables alive() check the -// design specifies is kept anyway, as a cheap early exit before -// starting a table's work rather than the only guard against loss of -// exclusivity. -// - F26 (release safety): release() unlocks and closes on an independent +// - Exclusivity window: because every sweep statement runs on the lock +// connection, there is no window where work proceeds on a different +// connection after the session died — if the session ends, the very +// next statement on it fails immediately instead of continuing to run +// against the pool while another backend has already re-acquired the +// lock. The between-tables alive() check the design specifies is kept +// anyway, as a cheap early exit before starting a table's work rather +// than the only guard against loss of exclusivity. +// - Release safety: release() unlocks and closes on an independent // context, not the sweep's (which may already be cancelled), with a // bounded timeout, and discards the connection via driver.ErrBadConn // whenever the unlock did not provably succeed — so database/sql @@ -76,7 +73,8 @@ func acquireSweepSession(ctx context.Context, pool *db.Pool) (*sweepSession, boo } // queryer is the connection every sweep statement must run through for the -// whole sweep's duration (F27/F25 above). +// whole sweep's duration (see the single-connection-budget and +// exclusivity-window invariants on sweepSession above). func (s *sweepSession) queryer() queryer { return s.conn } @@ -102,7 +100,7 @@ func (s *sweepSession) release() { // The unlock did not provably succeed: force database/sql to // discard this connection instead of returning it to the pool, // so a session that may still hold the lock can never be reused - // by a later, unrelated caller (F26). + // by a later, unrelated caller. _ = s.conn.Raw(func(driverConn any) error { return driver.ErrBadConn }) } _ = s.conn.Close() diff --git a/apps/backend/internal/office/retention/preview_marker.go b/apps/backend/internal/office/retention/preview_marker.go index dc668b8ef21..ef8942298f4 100644 --- a/apps/backend/internal/office/retention/preview_marker.go +++ b/apps/backend/internal/office/retention/preview_marker.go @@ -73,10 +73,12 @@ func (s *PreviewMarkerStore) MarkCompleted(ctx context.Context, table TableName, // GetWith is Get against an explicit connection instead of the shared // settings pool. Required mid-sweep on PostgreSQL: every statement in a -// sweep must run on the session holding the advisory lock (F27 — see -// sweep.go and lock.go), and going through the pool here would request a -// second connection, which deadlocks under a maxOpenConns=1 pool (the -// mandated Postgres-gated test harness). +// sweep must run on the session holding the advisory lock (see sweep.go +// and lock.go), and going through the pool here would request a second +// connection, which deadlocks under a maxOpenConns=1 pool (the mandated +// Postgres-gated test harness). This bypasses systemsettings.Store and +// reads the key/value pair directly, so it depends on that package's +// `settings` table keeping its `key`/`value` column names. func (s *PreviewMarkerStore) GetWith(ctx context.Context, q queryer) (PreviewMarker, bool) { var raw string err := q.GetContext(ctx, &raw, q.Rebind(`SELECT value FROM settings WHERE key = ?`), previewMarkerKey) diff --git a/apps/backend/internal/office/retention/settings_wire.go b/apps/backend/internal/office/retention/settings_wire.go index e82f76af7ba..4dff218b6f7 100644 --- a/apps/backend/internal/office/retention/settings_wire.go +++ b/apps/backend/internal/office/retention/settings_wire.go @@ -48,6 +48,9 @@ func decodeRetentionSettings(body []byte) (Settings, error) { if err := dec.Decode(&env); err != nil { return Settings{}, err } + if dec.More() { + return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object") + } enabled, err := decodeBoolField(env.Enabled, "enabled", defaults.Enabled) if err != nil { diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go index bcf3fe980b0..573a6892dbb 100644 --- a/apps/backend/internal/office/retention/store.go +++ b/apps/backend/internal/office/retention/store.go @@ -326,6 +326,9 @@ func deleteRunsByIDs(ctx context.Context, tx *sqlx.Tx, ids []string, cutoff time return total, nil } +// deleteByRunIDs deletes run-id-keyed satellite rows for one table. table +// must be a hardcoded identifier from a call site in this package, never a +// caller-supplied string — it is interpolated directly into the query text. func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) { var total int64 for _, chunk := range chunkIDs(ids, retentionMaxHostParams) { diff --git a/apps/backend/internal/office/retention/sweep.go b/apps/backend/internal/office/retention/sweep.go index 8c0c36a6fd2..9c972d19cf4 100644 --- a/apps/backend/internal/office/retention/sweep.go +++ b/apps/backend/internal/office/retention/sweep.go @@ -49,11 +49,10 @@ type satelliteResults struct { } // Sweeper runs one sweep or one census pass at a time; the scheduler -// (scheduler.go) owns when to call each. Findings resolved here, as -// documented Build decisions: +// (scheduler.go) owns when to call each. // -// - F28 (lost-exclusivity outcome): a lock lost between tables abandons -// the whole sweep attempt as a skip — the same outcome as a local +// - Lost-exclusivity outcome: a lock lost between tables abandons the +// whole sweep attempt as a skip — the same outcome as a local // concurrent-sweep collision — rather than inventing a per-table // "skipped" state the design's eight-id health catalogue has nowhere // to report. Batches already committed on PostgreSQL stay committed @@ -152,10 +151,10 @@ func (s *Sweeper) RunSweep(ctx context.Context) { testBetweenTablesSweep(q) } if session != nil && !session.alive(ctx) { - // F25/F28: exclusivity was lost after the first table's work. Do - // not start the second table, and do not publish a partial - // result — this whole attempt is a skip, exactly as if the lock - // had never been acquired. + // Exclusivity was lost after the first table's work. Do not + // start the second table, and do not publish a partial result — + // this whole attempt is a skip, exactly as if the lock had + // never been acquired. s.recordSkip() return } @@ -182,7 +181,7 @@ func (s *Sweeper) RunSweep(ctx context.Context) { // testBetweenTablesSweep, when set, runs right after office_routine_runs' // table work and right before the alive() liveness check and the runs // table — a deterministic seam for exercising AC-OFFICE-RUN-HISTORY-RETENTION-002.12's -// "verifies the lock connection is still alive between tables" path (F25) +// "verifies the lock connection is still alive between tables" path // without depending on real cross-process timing. It receives the sweep's // own queryer so a test can run diagnostics (or a second sweep attempt) // against the exact connection in use. Never set outside tests. @@ -190,7 +189,7 @@ var testBetweenTablesSweep func(q queryer) // RunCensus evaluates the retained-count census for every thresholded // table. Read-only, so it needs no advisory lock: every backend computes -// and serves its own local view (F35). +// and serves its own local view. func (s *Sweeper) RunCensus(ctx context.Context) { q := s.pool.Reader() now := time.Now().UTC() @@ -244,7 +243,7 @@ func (s *Sweeper) acquireQueryer(ctx context.Context) (queryer, *sweepSession, b } // isPreviewed reads the preview marker through q, the sweep's own -// connection, rather than the shared settings pool (F27 — see the +// connection, rather than the shared settings pool (see the // PreviewMarkerStore.GetWith doc comment). func (s *Sweeper) isPreviewed(ctx context.Context, q queryer, table TableName) bool { marker, _ := s.previewMarker.GetWith(ctx, q) diff --git a/apps/web/components/settings/system/retention-settings-card.test.tsx b/apps/web/components/settings/system/retention-settings-card.test.tsx index 29c61eeb7d9..b3bb2a6a1b5 100644 --- a/apps/web/components/settings/system/retention-settings-card.test.tsx +++ b/apps/web/components/settings/system/retention-settings-card.test.tsx @@ -98,29 +98,6 @@ describe("RetentionSettingsCard", () => { expect(screen.getByTestId("retention-never-swept")).toBeTruthy(); }); - it("stages an admin edit until the shared save contributor runs, then reloads", async () => { - renderCard(); - await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); - - const toggle = screen.getByTestId(ENABLED_TOGGLE_TEST_ID); - fireEvent.click(toggle); - expect(saveRetentionSettingsMock).not.toHaveBeenCalled(); - expect(saveContributor?.isDirty).toBe(true); - if (!saveContributor) throw new Error("expected save contributor"); - - saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); - fetchRetentionStatusMock.mockResolvedValueOnce( - statusOf({ settings: defaultSettings({ enabled: false }) }), - ); - - await act(async () => saveContributor?.save(saveContributor.revision)); - - expect(saveRetentionSettingsMock).toHaveBeenCalledWith( - expect.objectContaining({ enabled: false }), - ); - await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); - }); - it("keeps members read-only while preserving the loaded values", async () => { currentRole = "member"; renderCard(); @@ -195,6 +172,48 @@ describe("RetentionSettingsCard", () => { }); }); +describe("RetentionSettingsCard save/reload consistency", () => { + it("stages an admin edit until the shared save contributor runs, then reloads", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + + const toggle = screen.getByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(toggle); + expect(saveRetentionSettingsMock).not.toHaveBeenCalled(); + expect(saveContributor?.isDirty).toBe(true); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); + fetchRetentionStatusMock.mockResolvedValueOnce( + statusOf({ settings: defaultSettings({ enabled: false }) }), + ); + + await act(async () => saveContributor?.save(saveContributor.revision)); + + expect(saveRetentionSettingsMock).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + ); + await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); + }); + + it("clears the dirty draft from the save response even when the post-save reload fails", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); + fetchRetentionStatusMock.mockRejectedValueOnce(new Error("offline")); + + await act(async () => saveContributor?.save(saveContributor.revision)); + + expect(saveRetentionSettingsMock).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + ); + await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); + }); +}); + describe("RetentionSettingsCard unknown-status reporting", () => { it("renders each unrecognized status with its row count", async () => { fetchRetentionStatusMock.mockResolvedValue( diff --git a/apps/web/components/settings/system/system-route-copy.test.ts b/apps/web/components/settings/system/system-route-copy.test.ts index 669405ac22c..1024b5e15df 100644 --- a/apps/web/components/settings/system/system-route-copy.test.ts +++ b/apps/web/components/settings/system/system-route-copy.test.ts @@ -17,6 +17,7 @@ vi.mock("@/components/settings/settings-target", () => ({ vi.mock("./backups-table", () => ({ BackupsTable: () => null })); vi.mock("./database-stats-card", () => ({ DatabaseStatsCard: () => null })); vi.mock("./log-viewer", () => ({ LogViewer: () => null })); +vi.mock("./retention-settings-card", () => ({ RetentionSettingsCard: () => null })); afterEach(() => { cleanup(); databaseState.value = null; @@ -147,6 +148,7 @@ describe("Data & Logs composition", () => { render(createElement(DataLogsSettings)); expect(screen.getByText(t("system:navDatabase"))).toBeTruthy(); + expect(screen.getByText(t("system:navRetention"))).toBeTruthy(); expect(screen.getByText(t("system:navBackups"))).toBeTruthy(); expect(screen.getByText(t("system:navLogs"))).toBeTruthy(); expect(screen.queryByText(t("system:storageTitle"))).toBeNull(); diff --git a/apps/web/hooks/domains/system/use-retention-settings.ts b/apps/web/hooks/domains/system/use-retention-settings.ts index 18dd4d65ae4..d93f2d6deb6 100644 --- a/apps/web/hooks/domains/system/use-retention-settings.ts +++ b/apps/web/hooks/domains/system/use-retention-settings.ts @@ -1,13 +1,14 @@ "use client"; import { useCallback, useEffect, useState } from "react"; -import { useAppStore } from "@/components/state-provider"; +import { useAppStore, useAppStoreApi } from "@/components/state-provider"; import { fetchRetentionStatus, saveRetentionSettings } from "@/lib/api/domains/system-api"; import type { RetentionSettings } from "@/lib/types/system"; export function useRetentionSettings() { const status = useAppStore((s) => s.system.retention); const setStatus = useAppStore((s) => s.setSystemRetention); + const storeApi = useAppStoreApi(); const [isLoading, setIsLoading] = useState(false); const [error, setError] = useState(null); const [saveError, setSaveError] = useState(null); @@ -34,7 +35,24 @@ export function useRetentionSettings() { setSaveError(null); try { const saved = await saveRetentionSettings(settings); - await reload(); + // Apply the PUT's own normalized response synchronously, rather + // than relying solely on the reload() below: if that GET fails, + // its error is recorded but never surfaces once status is already + // loaded (see the isLoading/error-gated branches in + // RetentionSettingsCard), which would otherwise leave the store + // holding pre-save settings while the save coordinator believes + // the save already succeeded. + storeApi.setState((state) => + state.system.retention + ? { + system: { + ...state.system, + retention: { ...state.system.retention, settings: saved }, + }, + } + : state, + ); + void reload(); return saved; } catch (e) { const message = e instanceof Error ? e.message : String(e); @@ -42,7 +60,7 @@ export function useRetentionSettings() { throw e; } }, - [reload], + [reload, storeApi], ); return { status, isLoading, error, saveError, reload, save }; From 8dee3af3143c125836edee265a438524a13c2ea5 Mon Sep 17 00:00:00 2001 From: Carlos Florencio Date: Sun, 13 Sep 2026 04:26:56 +0100 Subject: [PATCH 6/8] fix(office): protect retention recovery state and sync settings Keep pause-recovery runs until consumed, reconcile shared settings during census, and harden request parsing. Document policy and gate PostgreSQL retention tests. --- .github/workflows/backend-tests.yml | 1 + .../repository/sqlite/base_migrations.go | 3 + .../sqlite/retention_indexes_postgres_test.go | 2 + .../sqlite/retention_indexes_test.go | 11 ++- .../internal/office/retention/handler.go | 11 +++ .../internal/office/retention/handler_test.go | 30 +++++++++ .../internal/office/retention/scheduler.go | 19 +++++- .../office/retention/scheduler_test.go | 26 +++++++ .../office/retention/settings_wire.go | 15 ++++- .../internal/office/retention/store.go | 7 +- .../internal/office/retention/store_test.go | 67 +++++++++++++++++++ docs/public/operations.md | 28 ++++++++ .../requirements/run-history-retention.md | 10 ++- .../run-history-retention-operations.md | 7 +- .../system-design/run-history-retention.md | 26 +++++-- 15 files changed, 243 insertions(+), 20 deletions(-) diff --git a/.github/workflows/backend-tests.yml b/.github/workflows/backend-tests.yml index e3d01d15396..1f7650c4711 100644 --- a/.github/workflows/backend-tests.yml +++ b/.github/workflows/backend-tests.yml @@ -624,6 +624,7 @@ jobs: ./internal/notifications/store ./internal/office/configsync ./internal/office/repository/sqlite + ./internal/office/retention ./internal/orchestrator/messagequeue ./internal/persistence ./internal/persistence/storeconformance diff --git a/apps/backend/internal/office/repository/sqlite/base_migrations.go b/apps/backend/internal/office/repository/sqlite/base_migrations.go index 5f7fca3f3a2..968b8e9e7af 100644 --- a/apps/backend/internal/office/repository/sqlite/base_migrations.go +++ b/apps/backend/internal/office/repository/sqlite/base_migrations.go @@ -442,6 +442,9 @@ func (r *Repository) migrateFailureColumns() error { if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_agent ON office_agent_pause_recoveries(agent_id)`); err != nil { return fmt.Errorf("idx_office_agent_pause_recoveries_agent: %w", err) } + if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)`); err != nil { + return fmt.Errorf("idx_office_agent_pause_recoveries_failed_run: %w", err) + } return nil } diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go index 35f3bd2362e..dde7028f0f5 100644 --- a/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go @@ -27,12 +27,14 @@ func TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { } assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") assertPostgresIndexExists(t, conn, "idx_runs_retention") + assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { t.Fatalf("replay NewWithDB: %v", err) } assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") assertPostgresIndexExists(t, conn, "idx_runs_retention") + assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") } func assertPostgresIndexExists(t *testing.T, conn interface { diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go index e95919f9134..1ee5fe893d2 100644 --- a/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go @@ -9,12 +9,9 @@ import ( "github.com/kandev/kandev/internal/office/repository/sqlite" ) -// TestRetentionIndexes_CreatedFreshAndReplaySafe proves the two indexes the -// retention sweep depends on (idx_office_routine_runs_retention, -// idx_runs_retention) exist after a fresh boot and that re-running schema -// init against the same database (the upgrade-path replay) is a no-op, not -// an error — CREATE INDEX IF NOT EXISTS over the same COALESCE(...) -// expression both paths must produce identically. +// TestRetentionIndexes_CreatedFreshAndReplaySafe proves the retention indexes +// exist after a fresh boot and that re-running schema init against the same +// database (the upgrade-path replay) is a no-op, not an error. func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { conn, err := sqlx.Open("sqlite3", ":memory:") if err != nil { @@ -28,6 +25,7 @@ func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { } assertIndexExists(t, conn, "idx_office_routine_runs_retention") assertIndexExists(t, conn, "idx_runs_retention") + assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") // Replay: schema init against the same, already-initialized database. if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { @@ -35,6 +33,7 @@ func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { } assertIndexExists(t, conn, "idx_office_routine_runs_retention") assertIndexExists(t, conn, "idx_runs_retention") + assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") } func assertIndexExists(t *testing.T, conn *sqlx.DB, name string) { diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go index 58fd21be233..d68b5e83826 100644 --- a/apps/backend/internal/office/retention/handler.go +++ b/apps/backend/internal/office/retention/handler.go @@ -11,6 +11,11 @@ import ( const responseErrorKey = "error" +// maxRetentionSettingsBodyBytes bounds administrator-controlled JSON before +// the handler decodes it, so a malformed request cannot consume unbounded +// memory. +const maxRetentionSettingsBodyBytes = 1 << 20 + // HandlerConfig wires the HTTP surface to the package's own stores. type HandlerConfig struct { SettingsStore *SettingsStore @@ -86,8 +91,14 @@ func (h *Handler) getRetention(c *gin.Context) { } func (h *Handler) putRetention(c *gin.Context) { + c.Request.Body = http.MaxBytesReader(c.Writer, c.Request.Body, maxRetentionSettingsBodyBytes) body, err := c.GetRawData() if err != nil { + var maxBytesErr *http.MaxBytesError + if errors.As(err, &maxBytesErr) { + c.JSON(http.StatusRequestEntityTooLarge, gin.H{responseErrorKey: "request body too large"}) + return + } c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: "failed to read request body"}) return } diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go index 29111c353b8..a8909ef9b7b 100644 --- a/apps/backend/internal/office/retention/handler_test.go +++ b/apps/backend/internal/office/retention/handler_test.go @@ -251,6 +251,36 @@ func TestPutRetention_TrailingDataAfterObjectIsRejected(t *testing.T) { } } +func TestPutRetention_TopLevelNullIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte("null")) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_OversizedBodyIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(strings.Repeat(" ", maxRetentionSettingsBodyBytes+1)) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("status = %d, want 413: %s", response.Code, response.Body.String()) + } +} + func TestPutRetention_FractionalNumberIsRejected(t *testing.T) { gin.SetMode(gin.TestMode) handler, _ := newTestHandler(t) diff --git a/apps/backend/internal/office/retention/scheduler.go b/apps/backend/internal/office/retention/scheduler.go index 35a035275fe..ec0725783eb 100644 --- a/apps/backend/internal/office/retention/scheduler.go +++ b/apps/backend/internal/office/retention/scheduler.go @@ -36,7 +36,8 @@ type SchedulerOptions struct { // moment of the change (fixed-delay, not fixed-rate): the sweep timer at // firstSweepDelay only when retention just turned on, otherwise at the // (possibly new) full interval; the census timer always at the full -// interval. +// interval. The census timer also refreshes the shared settings record, so a +// backend that did not serve a settings write still adopts it while disabled. type Scheduler struct { settingsStore *SettingsStore sweeper *Sweeper @@ -155,6 +156,22 @@ func (s *Scheduler) run(ctx context.Context, settings Settings, wake <-chan stru case <-census: s.sweeper.RunCensus(ctx) + if latest, err := s.settingsStore.GetSettings(ctx); err == nil && latest != settings { + wasEnabled := settings.Enabled + settings = latest + s.mu.Lock() + s.latest = latest + s.mu.Unlock() + + sweep = nil + if settings.Enabled { + if wasEnabled { + sweep = s.after(sweepInterval(settings)) + } else { + sweep = s.after(firstSweepDelay) + } + } + } census = s.after(sweepInterval(settings)) case <-sweep: diff --git a/apps/backend/internal/office/retention/scheduler_test.go b/apps/backend/internal/office/retention/scheduler_test.go index 2e9fb6ca3cd..b790497c5a9 100644 --- a/apps/backend/internal/office/retention/scheduler_test.go +++ b/apps/backend/internal/office/retention/scheduler_test.go @@ -231,6 +231,32 @@ func TestScheduler_DisablingStopsArmingSweepButNotCensus(t *testing.T) { fake.assertNotArmed(t, firstSweepDelay) } +func TestScheduler_ReconcilesSharedSettingsOnCensus(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + settings.SweepIntervalHours = 1 + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + updated := settings + updated.Enabled = true + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), updated); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + fake.fire(t, sweepInterval(settings)) + + // The census timer is also the periodic shared-settings reconciliation + // point. Enabling retention in another backend must arm this process's + // first sweep even when no local PUT delivered ApplySettings. + fake.waitArmed(t, firstSweepDelay) +} + func TestScheduler_StartTwiceIsNoop(t *testing.T) { fake := newFakeAfter() settings := DefaultSettings() diff --git a/apps/backend/internal/office/retention/settings_wire.go b/apps/backend/internal/office/retention/settings_wire.go index 4dff218b6f7..73af2563374 100644 --- a/apps/backend/internal/office/retention/settings_wire.go +++ b/apps/backend/internal/office/retention/settings_wire.go @@ -4,6 +4,7 @@ import ( "bytes" "encoding/json" "fmt" + "io" ) // retentionSettingsEnvelope captures each top-level field as raw JSON @@ -41,15 +42,23 @@ type runEventsSettingsEnvelope struct { // SettingsStore.SaveSettings after this decode succeeds. func decodeRetentionSettings(body []byte) (Settings, error) { defaults := DefaultSettings() + trimmed := bytes.TrimSpace(body) + if len(trimmed) == 0 || trimmed[0] != '{' { + return Settings{}, fmt.Errorf("request body: expected a JSON object") + } - dec := json.NewDecoder(bytes.NewReader(body)) + dec := json.NewDecoder(bytes.NewReader(trimmed)) dec.DisallowUnknownFields() var env retentionSettingsEnvelope if err := dec.Decode(&env); err != nil { return Settings{}, err } - if dec.More() { - return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object") + var extra any + if err := dec.Decode(&extra); err != io.EOF { + if err == nil { + return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object") + } + return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object: %w", err) } enabled, err := decodeBoolField(env.Enabled, "enabled", defaults.Enabled) diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go index 573a6892dbb..52d80b8b138 100644 --- a/apps/backend/internal/office/retention/store.go +++ b/apps/backend/internal/office/retention/store.go @@ -69,7 +69,12 @@ func runEligibleSubquery() string { ORDER BY COALESCE(finished_at, requested_at) DESC, id DESC ) AS rn FROM runs - WHERE status IN (?)` + WHERE status IN (?) + AND NOT EXISTS ( + SELECT 1 + FROM office_agent_pause_recoveries + WHERE office_agent_pause_recoveries.failed_run_id = runs.id + )` } // CountEligibleRoutineRuns is the office_routine_runs eligibility count, diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go index bb38bbfae10..3db5aed1918 100644 --- a/apps/backend/internal/office/retention/store_test.go +++ b/apps/backend/internal/office/retention/store_test.go @@ -74,6 +74,16 @@ func seedRun(t *testing.T, conn *sqlx.DB, id, agentProfileID, status string, fin } } +func seedPauseRecovery(t *testing.T, conn *sqlx.DB, agentID, taskID, failedRunID string) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_agent_pause_recoveries (agent_id, task_id, failed_run_id) + VALUES (?, ?, ?) + `), agentID, taskID, failedRunID); err != nil { + t.Fatalf("seed pause recovery for run %s: %v", failedRunID, err) + } +} + func seedRunEvent(t *testing.T, conn *sqlx.DB, runID string, seq int) { t.Helper() if _, err := conn.Exec(conn.Rebind(` @@ -301,6 +311,63 @@ func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { } } +// TestRunRetention_PreservesActivePauseRecoveryRun proves that a failed run +// still referenced by MarkAgentPausedFixed remains available until its +// recovery snapshot is consumed or discarded. Count and delete must use the +// same protection predicate so a preview cannot promise deletion that the +// batch path applies. +func TestRunRetention_PreservesActivePauseRecoveryRun(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + failedAt := daysAgo(60) + seedRun(t, conn, runID, "agent-recovery", "failed", &failedAt, failedAt) + seedPauseRecovery(t, conn, "agent-recovery", "task-recovery", runID) + + cutoff := daysAgo(30) + count, err := store.CountEligibleRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRuns: %v", err) + } + if count != 0 { + t.Fatalf("eligible count = %d, want 0 while pause recovery references the run", count) + } + + result, err := store.DeleteRunBatch(ctx, conn, cutoff, 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op while recovery is active", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("recovery run count = %d, want 1", n) + } + + if _, err := conn.Exec(conn.Rebind( + `DELETE FROM office_agent_pause_recoveries WHERE agent_id = ? AND task_id = ?`, + ), "agent-recovery", "task-recovery"); err != nil { + t.Fatalf("discard pause recovery: %v", err) + } + + count, err = store.CountEligibleRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRuns after recovery discard: %v", err) + } + if count != 1 { + t.Fatalf("eligible count after recovery discard = %d, want 1", count) + } + result, err = store.DeleteRunBatch(ctx, conn, cutoff, 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch after recovery discard: %v", err) + } + if result.RunsDeleted != 1 || result.Abandoned { + t.Fatalf("result after recovery discard = %+v, want one deleted run", result) + } +} + // TestDeleteRunBatch_SurvivingRunKeepsEveryEvent proves // AC-OFFICE-RUN-HISTORY-RETENTION-001.6: run_events is never deleted for a // run that is not itself being deleted in the same transaction, no matter diff --git a/docs/public/operations.md b/docs/public/operations.md index 7ce3be85abd..58d19172a06 100644 --- a/docs/public/operations.md +++ b/docs/public/operations.md @@ -348,6 +348,34 @@ the Kandev service first provides the clearest maintenance boundary. +## Office run history retention + +Open **Settings > System > Data & Logs** to manage automatic deletion of old +Office run history. Deletion is enabled by default. The first sweep starts five +minutes after the backend starts or after you enable deletion. + +The first sweep for each history table is a preview. It reports the rows that +would be deleted and removes no rows. A later sweep can delete eligible rows. +Deletion is permanent. Back up the database before you enable deletion if you +need to keep old history outside the configured window. + +The retention window controls the age of rows that can be deleted. The minimum +kept per owner control keeps the newest rows for each routine or agent, even +when those rows are older than the window. A floor of zero removes this extra +protection. Run event, route attempt, and skill rows are deleted with their +parent run. + +The page shows the current policy, retained row counts, preview results, the +last sweep, and any backlog or errors. Counts continue to update when deletion +is disabled, so you can monitor growth before you enable it again. + +To disable automatic deletion: + +1. Open **Settings > System > Data & Logs**. +2. Clear **Delete eligible run history**. +3. Select **Save changes**. +4. Check the retention status. It must show that deletion is disabled. + ## Database operation > **Single-owner rule:** SQLite uses one writer connection in WAL mode; only one Kandev backend should own the file. **Factory reset** is destructive and removes managed data after creating a pre-reset backup. diff --git a/docs/specs/office/requirements/run-history-retention.md b/docs/specs/office/requirements/run-history-retention.md index 2956b949518..535a1bf9d04 100644 --- a/docs/specs/office/requirements/run-history-retention.md +++ b/docs/specs/office/requirements/run-history-retention.md @@ -47,6 +47,9 @@ separate capability that never touches database rows. live decision reads. History rows are eligible for deletion. - **Live-state row**: a row a live decision still reads, regardless of age. Never eligible for age-based deletion. +- **Recovery-protected run**: a failed `runs` row named by an active + `office_agent_pause_recoveries.failed_run_id`; retention keeps it until the + recovery row is consumed or discarded. - **Run satellite row**: a row keyed by a `runs` row's identifier and owned by it: a `run_events`, `office_run_route_attempts`, or `office_run_skills` entry. - **Retention window**: the age past which a history row becomes eligible, @@ -94,7 +97,9 @@ automatically, so a scheduled routine does not grow the database without limit. completion timestamp in the same statement. When a `runs` row's status is `queued` or `claimed`, the system shall treat it as a live-state row and shall not delete it on age, at any age, including a run parked for a future routing - retry. + retry. A failed run referenced by an active + `office_agent_pause_recoveries.failed_run_id` is also live state for + retention and shall remain until that recovery row is consumed or discarded. - **AC-OFFICE-RUN-HISTORY-RETENTION-001.3:** When a history row's completion time is older than that table's retention window, the system shall make it eligible for deletion. Completion time is defined in Terminology. A row's @@ -223,6 +228,9 @@ behavior. second one. A settings change re-arms the delay from the moment of the change. Enabling retention that was disabled arms the next sweep at the same short delay as AC-OFFICE-RUN-HISTORY-RETENTION-002.10 rather than at a full interval. + Each backend shall periodically reread the shared settings record, including + while retention is disabled, so a backend that did not serve a settings write + still adopts enablement and interval changes. ### REQ-OFFICE-RUN-HISTORY-RETENTION-005: Integrity and database engine parity diff --git a/docs/specs/office/system-design/run-history-retention-operations.md b/docs/specs/office/system-design/run-history-retention-operations.md index bcc04167250..f4195860f7c 100644 --- a/docs/specs/office/system-design/run-history-retention-operations.md +++ b/docs/specs/office/system-design/run-history-retention-operations.md @@ -33,7 +33,7 @@ Adjacent contracts read and constrained but not owned: ### Settings -One `system_settings` key, `office_run_retention`, holding a JSON document, +One `settings` key, `office_run_retention`, holding a JSON document, read through a `Get`/`Save` pair with `Normalize` on both, matching `internal/system/storage/settings.go`. Unparseable content returns the defaults plus a sentinel error the caller turns into a health issue, never a boot failure @@ -85,7 +85,7 @@ reading them to *delete by* fails closed. ### The preview marker -A second `system_settings` key, `office_run_retention_preview_completed`, +A second `settings` key, `office_run_retention_preview_completed`, holding a JSON object keyed by swept table name whose values are the timestamp at which that table's preview completed: `{"office_routine_runs": "...", "runs": "..."}`. @@ -323,6 +323,9 @@ Unit tests in `internal/office/retention`, plus the frontend checks below. full eligible count and raises no backlog warning (003.6, 003.9). - A settings change written through one store handle is used by a sweep driven from a second handle that never saw the change notification (004.5). +- A backend census refresh adopts a settings change written by another backend, + including when retention was disabled, and arms the appropriate sweep timer + (002.13, 004.5). - A sweep whose settings read fails is skipped and recorded as skipped, and deletes nothing under the default window (004.5). - A `runs` batch abandoned after its retry reports zero deleted for the three diff --git a/docs/specs/office/system-design/run-history-retention.md b/docs/specs/office/system-design/run-history-retention.md index 31868ae5def..c0831c0ad1d 100644 --- a/docs/specs/office/system-design/run-history-retention.md +++ b/docs/specs/office/system-design/run-history-retention.md @@ -78,6 +78,10 @@ Verified by reading the schema owners, not assumed. transaction that deletes the run. - `runs` indexes are `idx_run_status_requested (status, requested_at)` and the partial unique `idx_run_idempotency`. Nothing indexes `finished_at`. +- `office_agent_pause_recoveries.failed_run_id` identifies failed runs that an + active pause recovery still needs. It has no foreign key, so retention uses a + correlated `NOT EXISTS` predicate and the supporting + `idx_office_agent_pause_recoveries_failed_run` index. - `ScheduleRetry` (`runs/repository/sqlite/runs.go`) sets `status = 'queued', finished_at = NULL` on an existing run, keyed by id with **no status guard in its `WHERE` clause**. Any terminal run can therefore @@ -187,7 +191,7 @@ with participant-seat, secret-transfer, or workflow-phase locking. **If the lock connection drops mid-sweep**, PostgreSQL releases the session's advisory locks as part of ending the session, so a crashed or partitioned backend cannot wedge retention permanently — that self-healing is the reason for a -session lock rather than a lease row in `system_settings`, which would need its +session lock rather than a lease row in the `settings` table, which would need its own expiry and its own stale-holder rule. The cost is that the surviving sweep no longer holds exclusivity without knowing it, so the sweep **verifies the lock connection is still alive between tables** and, if it is not, stops before the @@ -223,6 +227,10 @@ as a fresh start rather than at a full interval — otherwise an operator who enables retention on a 168-hour interval waits a week to see whether it works (AC-OFFICE-RUN-HISTORY-RETENTION-002.13). +The census timer also re-reads shared settings. It runs while retention is +disabled, so other backends discover enablement and interval changes and re-arm +their timers. + ### Eligibility, expressed once Two predicates, each defined in exactly one place and reused by the count, the @@ -288,7 +296,11 @@ database. **Runs.** History is `status IN ('finished','failed','cancelled')`, partitioned by `agent_profile_id`, ranked `COALESCE(finished_at, requested_at) DESC, id DESC` -and batch-ordered `COALESCE(finished_at, requested_at) ASC, id ASC`. There is no +and batch-ordered `COALESCE(finished_at, requested_at) ASC, id ASC`. A failed +run named by an active `office_agent_pause_recoveries.failed_run_id` is +protected by a `NOT EXISTS` clause in this predicate. Count, selection, and +delete use the same clause, so a preview cannot promise deletion of a run that +the recovery flow still needs. There is no `finished_at IS NOT NULL` conjunct: requiring one would make a terminal row with an unset stamp immortal and unobservable, which AC-OFFICE-RUN-HISTORY-RETENTION-001.3 forbids. `queued` and `claimed` are @@ -340,7 +352,8 @@ polls; an unordered list would make a stable condition look like a changing one. Per batch, one transaction, satellites first, run last: -1. Select up to `batch_limit` eligible run ids. +1. Select up to `batch_limit` eligible run ids, excluding runs named by an + active `office_agent_pause_recoveries.failed_run_id`. 2. `DELETE FROM run_events WHERE run_id IN (...)` 3. `DELETE FROM office_run_route_attempts WHERE run_id IN (...)` 4. `DELETE FROM office_run_skills WHERE run_id IN (...)` @@ -382,10 +395,11 @@ id set (AC-OFFICE-RUN-HISTORY-RETENTION-001.6). There is no age-based delete on ### Indexes to add -Neither table has an index serving the sweep. +Retention adds indexes that serve the sweep. -- `idx_office_routine_runs_retention ON office_routine_runs(routine_id, completed_at DESC, id DESC)` -- `idx_runs_retention ON runs(agent_profile_id, finished_at DESC, id DESC)` +- `idx_office_routine_runs_retention ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)` +- `idx_runs_retention ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)` +- `idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)` Added through the existing `office/repository/sqlite` schema path so both the fresh-install `CREATE` and the upgrade path get them, and recorded in the From 111f02840e294033c07a3f404f7037bc172ebccb Mon Sep 17 00:00:00 2001 From: nova28 <17953305+nova28@users.noreply.github.com> Date: Sun, 13 Sep 2026 12:11:13 +0800 Subject: [PATCH 7/8] docs(office): add the retention work order the docs-coverage gate needs The base branch's new "PR documentation coverage" check (introduced after this branch was opened) requires every changed non-exempt path to be covered by a linked docs/plans//task-NN-*.md work order with a sibling plan.md, both referencing the feature's existing frozen spec. Add that pair for the already-implemented run-history-retention work so the check passes. Co-Authored-By: Claude Sonnet 5 --- docs/plans/run-history-retention/plan.md | 64 ++++++ ...-bound-run-history-with-retention-sweep.md | 183 ++++++++++++++++++ 2 files changed, 247 insertions(+) create mode 100644 docs/plans/run-history-retention/plan.md create mode 100644 docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md diff --git a/docs/plans/run-history-retention/plan.md b/docs/plans/run-history-retention/plan.md new file mode 100644 index 00000000000..36b58620542 --- /dev/null +++ b/docs/plans/run-history-retention/plan.md @@ -0,0 +1,64 @@ +--- +created: 2026-09-09 +status: complete +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +system_design: + - ../../specs/office/system-design/run-history-retention.md + - ../../specs/office/system-design/run-history-retention-operations.md +legacy_specs: [] +--- + +# Implementation Plan: Office Run History Retention + +## Overview + +Office writes `office_routine_runs` and `run_events` rows on every routine firing and +run lifecycle transition, and nothing ever deletes them on a schedule. A `*/5 * * * *` +routine produces roughly 105,000 `office_routine_runs` rows a year; one reference +install had already accumulated 323 consecutive `coalesced` rows from a routine that +was doing nothing useful. This plan adds one scheduled sweep, on its own interval +(never the 5s Office tick), that bounds `office_routine_runs`, `runs`, and their +satellites (`run_events`, `office_run_route_attempts`, `office_run_skills`) by age, +with a per-owner floor, identical behavior on SQLite and PostgreSQL, and an operator +surface (Settings > System > Data & Logs) that reports policy, counts, previews, and +warnings before and while rows are deleted. + +The two halves of the contract are split the same way the specs are split: the sweep +itself, its eligibility rules, and engine parity are +[run history retention](../../specs/office/system-design/run-history-retention.md) +(REQ-001, REQ-002, REQ-005); the settings record, preview marker, health warnings, and +System page surface are +[run history retention operations](../../specs/office/system-design/run-history-retention-operations.md) +(REQ-003, REQ-004). + +## Scope + +### In scope + +- A `internal/office/retention` package: settings store, eligibility/count/delete + queries shared by preview and delete, a session-scoped PostgreSQL advisory lock with + a SQLite single-process equivalent, a scheduler goroutine on its own interval, and an + HTTP handler for `GET`/`PUT /api/v1/system/retention`. +- Status-only classification of history vs. live state for both `office_routine_runs` + and `runs`, a per-owner floor, oldest-first chunked batch deletion, and satellite rows + deleted in the same transaction as their parent run. +- A per-table preview (report, delete nothing) on each table's first evaluation, and + `health.Issue` warnings before the cap and on sweep/count failure. +- A `RetentionSettingsCard` on Settings > System > Data & Logs showing policy, retained + counts, preview state, last sweep, and backlog/error warnings. +- Expression indexes serving the sweep's filter/order on both engines. + +### Out of scope + +- Filesystem/container cleanup (owned by storage maintenance). +- Routine or workspace deletion (already deletes runs structurally; unaffected). +- Any change to run lifecycle, routine dispatch, or task/session/checkout data. + +## Tasks + +- [x] [Task 01: Bound run history with a scheduled retention sweep](task-01-bound-run-history-with-retention-sweep.md) diff --git a/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md new file mode 100644 index 00000000000..7d7c2c15465 --- /dev/null +++ b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md @@ -0,0 +1,183 @@ +--- +id: "01-bound-run-history-with-retention-sweep" +title: "Bound Office run history with a scheduled retention sweep" +status: done +wave: 1 +depends_on: [] +plan: "plan.md" +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +acceptance_criteria: + - AC-OFFICE-RUN-HISTORY-RETENTION-001.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.11 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.12 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.13 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.11 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.5 +system_design: + - ../../specs/office/system-design/run-history-retention.md + - ../../specs/office/system-design/run-history-retention-operations.md +--- + +# Task 01: Bound Office Run History with a Scheduled Retention Sweep + +## Summary + +Add `internal/office/retention`: a settings-backed sweep that ages out +`office_routine_runs` and `runs` history (plus their satellites) on its own +interval, a per-owner floor, status-only live/history classification, a preview +pass before first deletion, engine-identical behavior on SQLite and PostgreSQL, +and a `GET`/`PUT /api/v1/system/retention` operator surface with a matching +Settings > System > Data & Logs card. + +## In scope + +- Eligibility, count, and delete queries shared by preview and the real sweep, + keyed on `COALESCE(completed_at, created_at)` / `COALESCE(finished_at, + requested_at)`, oldest-first, chunked across statements. +- A session-scoped PostgreSQL advisory lock (with a SQLite single-process + equivalent) so only one backend sweeps at a time. +- A per-owner floor (newest N rows retained regardless of age) applied to + count, preview, and delete identically, re-asserted in the DELETE statement. +- Satellite deletion (`run_events`, `office_run_route_attempts`, + `office_run_skills`) in the same transaction as their parent run. +- `health.Issue` warnings for approaching the backlog cap and for count/sweep + failure; a one-time preview per table before any row is deleted. +- `RetentionSettingsCard` on Settings > System > Data & Logs: policy, retained + counts, preview/backlog state, last sweep, errors. +- Expression indexes serving the sweep's filter/order on both engines. + +## Out of scope + +- Filesystem/container artifact cleanup (storage maintenance owns that). +- Routine/workspace deletion cascades (already delete runs structurally). +- Any run lifecycle, routing, or task/session/checkout behavior change. + +## Acceptance + +- History rows (terminal `office_routine_runs`/`runs` statuses) older than the + configured window are deleted in batches; live-state rows (`received`, + `task_created`, `queued`, `claimed`) are never age-pruned; an unrecognized + status fails safe as live state and warns. +- The newest `floor` rows per owner survive regardless of age. +- Each table's first sweep is a preview: it reports would-delete counts and + deletes nothing; a later sweep performs real deletion. +- Behavior, including the advisory lock and batch chunking, is identical on + SQLite and PostgreSQL. +- `GET`/`PUT /api/v1/system/retention` read/write policy and report current + counts, preview state, last sweep outcome, and backlog/error warnings. + +## Verification + +```bash +cd apps/backend && go test -race -count=1 ./internal/office/retention/... ./internal/office/repository/sqlite/... +cd apps/backend && KANDEV_TEST_POSTGRES_DSN= go test -race -count=1 -v ./internal/office/retention/... +cd apps/backend && golangci-lint run ./internal/office/retention/... +cd apps/web && pnpm run typecheck +cd apps && pnpm --filter @kandev/web test -- --run retention-settings-card system-route-copy +cd apps/web && pnpm e2e:run -- --project chromium --grep "System retention settings" +``` + +## Files likely touched + +- `apps/backend/internal/office/retention/*.go` +- `apps/backend/internal/office/repository/sqlite/base_migrations.go`, + `retention_indexes_test.go`, `retention_indexes_postgres_test.go` +- `apps/backend/internal/backendapp/*` (scheduler/handler wiring) +- `apps/web/src/**/retention-settings-card.tsx` and its tests +- `docs/specs/office/requirements/run-history-retention*.md`, + `docs/specs/office/system-design/run-history-retention*.md` +- `docs/public/operations.md` + +## Dependencies + +None. + +## Risks + +- A dialect-sensitive query bug that only reproduces on PostgreSQL (mitigated + by an environment-gated Postgres suite covering the advisory lock, the + EvalPlanQual TOCTOU window, and batch chunking). +- A too-aggressive window or floor deleting rows an operator still needed + (mitigated by the default 30-day window, the per-owner floor, and the + preview-before-delete behavior). + +## Parallelism + +`sequential` + +## Inputs + +- `REQ-OFFICE-RUN-HISTORY-RETENTION-001` through `-005`. +- [run history retention](../../specs/office/system-design/run-history-retention.md) + and + [run history retention operations](../../specs/office/system-design/run-history-retention-operations.md). +- The reference install's 323 consecutive `coalesced` routine-run rows (28 + days, one bricked routine) cited in the requirements' Overview. + +## Results + +- Implemented `internal/office/retention` (settings store, `Sweeper`, + `Scheduler`, `CensusTracker`, PostgreSQL advisory lock with SQLite + equivalent, `Handler` for `GET`/`PUT /api/v1/system/retention`) plus the + `RetentionSettingsCard` on Settings > System > Data & Logs. +- Full backend gauntlet green: `go build ./...`, `go vet`, `gofmt -l`, + `go test -race ./internal/office/retention/...` (SQLite), the + PostgreSQL-gated suite against a real scratch instance (125 tests, 0 + skipped, all PASS, covering the advisory lock, the EvalPlanQual + concurrent-resurrection TOCTOU regression, and chunked batch deletes), + `golangci-lint run ./...` (0 issues). +- Frontend: `pnpm run typecheck`, the retention card and System route-copy + Vitest suites, and a scoped Playwright run (`--grep "System retention + settings"`, 2 passed). +- Delivered as PR [#3566](https://github.com/kdlbs/kandev/pull/3566) + ("feat(office): bound run history growth with a scheduled retention + sweep"), through four Build rounds and four Review rounds; remaining + non-blocking test-rigor gaps and CI-wiring follow-ups are tracked on the + linked follow-up card rather than blocking this PR. From 58c6918d276b81d7bf91a86117b500af5a71ef0b Mon Sep 17 00:00:00 2001 From: Carlos Florencio Date: Sun, 13 Sep 2026 17:30:48 +0100 Subject: [PATCH 8/8] test(office): align queue callers and stabilize failure tests --- .../lifecycle/manager_subscription_test.go | 3 ++- ...subscribers_lost_race_side_effects_test.go | 22 +++++++++++++++---- 2 files changed, 20 insertions(+), 5 deletions(-) diff --git a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go index d885a2fcaf3..f1ad97c6317 100644 --- a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go +++ b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go @@ -269,17 +269,18 @@ func TestAggregator_FailedPausedPushIsRetriedOnUnchangedContribution(t *testing. w.WriteHeader(http.StatusBadRequest) return } - modes <- body.Mode if body.Mode == string(WorkspacePollModePaused) { mu.Lock() pausedCalls++ call := pausedCalls mu.Unlock() if call == 1 { + modes <- body.Mode w.WriteHeader(http.StatusServiceUnavailable) return } } + modes <- body.Mode w.WriteHeader(http.StatusOK) })) t.Cleanup(srv.Close) diff --git a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go index a77e43d3c18..7952730f079 100644 --- a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go +++ b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go @@ -47,9 +47,17 @@ func TestHandleAgentCompleted_FinishRunFailureSkipsCompletionSideEffects(t *test // Force FinishRun's UPDATE to fail with a genuine error, matching // TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails' - // fault injection: a targeted fault (drop the column FinishRun sets) - // rather than a global read-only pragma. - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + // targeted fault rather than a global read-only pragma. + // Block only the terminal timestamp update. The retention expression index + // references finished_at, so dropping the column would fail before the + // handler runs and would no longer exercise its guarded error path. + svc.ExecSQL(t, ` + CREATE TRIGGER block_completion_finish_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "task_id": taskID, @@ -114,7 +122,13 @@ func TestHandleTasklessAgentCompleted_FinishRunFailureSkipsCompletionSideEffects ) `, agent.ID) - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + svc.ExecSQL(t, ` + CREATE TRIGGER block_taskless_completion_finish_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "agent_id": agent.ID,