diff --git a/.github/workflows/backend-tests.yml b/.github/workflows/backend-tests.yml index e3d01d15396..1f7650c4711 100644 --- a/.github/workflows/backend-tests.yml +++ b/.github/workflows/backend-tests.yml @@ -624,6 +624,7 @@ jobs: ./internal/notifications/store ./internal/office/configsync ./internal/office/repository/sqlite + ./internal/office/retention ./internal/orchestrator/messagequeue ./internal/persistence ./internal/persistence/storeconformance diff --git a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go index d885a2fcaf3..f1ad97c6317 100644 --- a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go +++ b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go @@ -269,17 +269,18 @@ func TestAggregator_FailedPausedPushIsRetriedOnUnchangedContribution(t *testing. w.WriteHeader(http.StatusBadRequest) return } - modes <- body.Mode if body.Mode == string(WorkspacePollModePaused) { mu.Lock() pausedCalls++ call := pausedCalls mu.Unlock() if call == 1 { + modes <- body.Mode w.WriteHeader(http.StatusServiceUnavailable) return } } + modes <- body.Mode w.WriteHeader(http.StatusOK) })) t.Cleanup(srv.Close) diff --git a/apps/backend/internal/backendapp/helpers.go b/apps/backend/internal/backendapp/helpers.go index 859ad57c127..d06ea408750 100644 --- a/apps/backend/internal/backendapp/helpers.go +++ b/apps/backend/internal/backendapp/helpers.go @@ -65,6 +65,7 @@ import ( notificationhandlers "github.com/kandev/kandev/internal/notifications/handlers" officeagents "github.com/kandev/kandev/internal/office/agents" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officetestharness "github.com/kandev/kandev/internal/office/testharness" "github.com/kandev/kandev/internal/orchestrator" "github.com/kandev/kandev/internal/org" @@ -1521,6 +1522,7 @@ func registerSecondaryRoutes( registerHealthRoutes(p) registerSystemRoutes(p) + registerRetentionRoutes(p) if p.runtimeFlagsSvc != nil { runtimeflags.RegisterRoutes(p.router, p.runtimeFlagsSvc) } @@ -1722,6 +1724,23 @@ func registerSystemRoutes(p routeParams) { p.systemSvc.RegisterRoutes(p.router, p.log) } +// registerRetentionRoutes mounts GET/PUT /api/v1/system/retention. It is a +// separate group from systemSvc's own /api/v1/system group (rather than a +// field on system.Service) because internal/office/retention cannot be +// imported by internal/system without inverting the existing system -> +// office dependency direction; gin allows two RouterGroups to share a path +// prefix as long as no route collides, and none does here. Read/admin +// split mirrors system.Service.RegisterRoutes: GET is member-readable, +// PUT requires the admin-scoped settings-manage permission. +func registerRetentionRoutes(p routeParams) { + if p.services == nil || p.services.Retention == nil { + return + } + read := p.router.Group("/api/v1/system") + admin := read.Group("", authz.RequireOrgScope(authz.ScopeOrgSettingsManage)) + retention.RegisterRoutes(read, admin, p.services.Retention.Handler) +} + // registerHealthRoutes sets up the system health endpoint with all health checkers. func registerHealthRoutes(p routeParams) { var githubProvider health.GitHubStatusProvider @@ -1747,6 +1766,9 @@ func registerHealthRoutes(p routeParams) { if p.systemSvc != nil && p.systemSvc.StorageRuntime != nil { checkers = append(checkers, p.systemSvc.StorageRuntime) } + if p.services != nil && p.services.Retention != nil { + checkers = append(checkers, p.services.Retention.Checker) + } healthSvc := health.NewService(p.log, checkers...) health.RegisterRoutes(p.router, healthSvc, p.log) } diff --git a/apps/backend/internal/backendapp/main.go b/apps/backend/internal/backendapp/main.go index 845bc80d67f..7ef9ad79846 100644 --- a/apps/backend/internal/backendapp/main.go +++ b/apps/backend/internal/backendapp/main.go @@ -100,6 +100,7 @@ import ( officepause "github.com/kandev/kandev/internal/office/pause" officeprojects "github.com/kandev/kandev/internal/office/projects" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officeroutines "github.com/kandev/kandev/internal/office/routines" "github.com/kandev/kandev/internal/office/routing" officescheduler "github.com/kandev/kandev/internal/office/scheduler" @@ -1181,6 +1182,16 @@ func startGatewayAndServe( }) systemSvc.Storage = storageComposition.handler systemSvc.StorageRuntime = storageComposition.runtime + + // Office run history retention: bounds office_routine_runs, runs, and + // their satellites on its own interval, separate from the 5s Office + // tick. Kept regardless of the Office feature flag — see Services.Retention. + services.Retention = retention.NewRuntime(dbPool, repos.SystemSettings, + func(message string, err error) { log.Error(message, zap.Error(err)) }) + if err := services.Retention.Start(ctx); err != nil { + log.Warn("office run retention scheduler failed to start", zap.Error(err)) + } + addCleanup(func() error { services.Retention.Stop(); return nil }) if systemSvc.LogBundles != nil { systemSvc.LogBundles.SetNotifier(gateway.Hub) systemSvc.LogBundles.SetSessionProvider(newDiagnosticSessionProvider(services.Task)) diff --git a/apps/backend/internal/backendapp/types.go b/apps/backend/internal/backendapp/types.go index 44107a9285f..84878a81fd7 100644 --- a/apps/backend/internal/backendapp/types.go +++ b/apps/backend/internal/backendapp/types.go @@ -25,6 +25,7 @@ import ( notificationstore "github.com/kandev/kandev/internal/notifications/store" office "github.com/kandev/kandev/internal/office" officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + "github.com/kandev/kandev/internal/office/retention" officeservice "github.com/kandev/kandev/internal/office/service" "github.com/kandev/kandev/internal/org" "github.com/kandev/kandev/internal/orgunit" @@ -112,6 +113,12 @@ type Services struct { // WorktreeMgr is the worktree manager. Exposed here so the install-wide // storage-maintenance composition can reach it for workspace cleanup. WorktreeMgr *worktree.Manager + // Retention owns the office_routine_runs/runs history sweep scheduler, + // its HTTP surface, and its health checker. Kept regardless of the + // Office feature flag, matching every other required-schema owner: rows + // written while Office was enabled still need bounding after it is + // turned off. + Retention *retention.Runtime // Terminal is the first-class user-terminal service (rename, park, etc.). // Wired into the gateway once lifecycle.Manager is up so the PTY backend // is available. diff --git a/apps/backend/internal/office/repository/sqlite/base_migrations.go b/apps/backend/internal/office/repository/sqlite/base_migrations.go index 75f2b6c6e99..968b8e9e7af 100644 --- a/apps/backend/internal/office/repository/sqlite/base_migrations.go +++ b/apps/backend/internal/office/repository/sqlite/base_migrations.go @@ -66,6 +66,9 @@ func (r *Repository) runMigrations() error { r.migrateBudgetPolicyRevision() r.migrateWorkspacePauseSkipAttribution() r.migrateLoopLivenessCausationID() + if err := r.migrateRetentionIndexes(); err != nil { + return err + } if err := r.migrate.Err(); err != nil { return err } @@ -141,6 +144,27 @@ func (r *Repository) migrateLoopLivenessCausationID() { ON runs(causation_id) WHERE causation_id != ''`) } +// migrateRetentionIndexes adds the two indexes the run-history retention +// sweep depends on (docs/specs/office/system-design/run-history-retention.md +// "Indexes to add"). Both are expression indexes over the same +// COALESCE(...) the sweep both filters and orders by; a plain-column index +// on the nullable completion column would serve neither the WHERE clause +// nor the ORDER BY the sweep actually issues, on either engine. +func (r *Repository) migrateRetentionIndexes() error { + if err := r.migrate.Apply( + "idx_office_routine_runs_retention", + `CREATE INDEX IF NOT EXISTS idx_office_routine_runs_retention + ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)`, + ); err != nil { + return err + } + return r.migrate.Apply( + "idx_runs_retention", + `CREATE INDEX IF NOT EXISTS idx_runs_retention + ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)`, + ) +} + // migrateContinuationScope adds runs.continuation_scope for databases // created before WO-16's claim-time scope persistence. Existing rows receive // a scope from their stored context snapshot so queued or claimed taskless @@ -418,6 +442,9 @@ func (r *Repository) migrateFailureColumns() error { if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_agent ON office_agent_pause_recoveries(agent_id)`); err != nil { return fmt.Errorf("idx_office_agent_pause_recoveries_agent: %w", err) } + if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)`); err != nil { + return fmt.Errorf("idx_office_agent_pause_recoveries_failed_run: %w", err) + } return nil } diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go new file mode 100644 index 00000000000..dde7028f0f5 --- /dev/null +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go @@ -0,0 +1,53 @@ +package sqlite_test + +import ( + "testing" + + "github.com/kandev/kandev/internal/office/repository/sqlite" + taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite" + "github.com/kandev/kandev/internal/testutil" +) + +// TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe is the PostgreSQL +// half of TestRetentionIndexes_CreatedFreshAndReplaySafe: the two +// expression indexes the retention sweep depends on must exist there too, +// with identical CREATE INDEX IF NOT EXISTS replay safety. Skips unless +// KANDEV_TEST_POSTGRES_DSN is set. +func TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + conn := testutil.OpenIsolatedPostgres(t, dsn) + + // tasks is created by the task repository's schema init, mirroring + // production boot order (see child_summaries_postgres_test.go). + if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init task repo: %v", err) + } + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("fresh NewWithDB: %v", err) + } + assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") + assertPostgresIndexExists(t, conn, "idx_runs_retention") + assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") + + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("replay NewWithDB: %v", err) + } + assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention") + assertPostgresIndexExists(t, conn, "idx_runs_retention") + assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") +} + +func assertPostgresIndexExists(t *testing.T, conn interface { + Get(dest interface{}, query string, args ...interface{}) error +}, name string) { + t.Helper() + var count int + if err := conn.Get(&count, + `SELECT COUNT(*) FROM pg_indexes WHERE indexname = $1`, name, + ); err != nil { + t.Fatalf("query pg_indexes for %s: %v", name, err) + } + if count != 1 { + t.Fatalf("index %s: found %d, want 1", name, count) + } +} diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go new file mode 100644 index 00000000000..1ee5fe893d2 --- /dev/null +++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go @@ -0,0 +1,50 @@ +package sqlite_test + +import ( + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// TestRetentionIndexes_CreatedFreshAndReplaySafe proves the retention indexes +// exist after a fresh boot and that re-running schema init against the same +// database (the upgrade-path replay) is a no-op, not an error. +func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) { + conn, err := sqlx.Open("sqlite3", ":memory:") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("fresh NewWithDB: %v", err) + } + assertIndexExists(t, conn, "idx_office_routine_runs_retention") + assertIndexExists(t, conn, "idx_runs_retention") + assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") + + // Replay: schema init against the same, already-initialized database. + if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("replay NewWithDB: %v", err) + } + assertIndexExists(t, conn, "idx_office_routine_runs_retention") + assertIndexExists(t, conn, "idx_runs_retention") + assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run") +} + +func assertIndexExists(t *testing.T, conn *sqlx.DB, name string) { + t.Helper() + var count int + if err := conn.Get(&count, + `SELECT COUNT(*) FROM sqlite_master WHERE type = 'index' AND name = ?`, name, + ); err != nil { + t.Fatalf("query sqlite_master for %s: %v", name, err) + } + if count != 1 { + t.Fatalf("index %s: found %d, want 1", name, count) + } +} diff --git a/apps/backend/internal/office/retention/census.go b/apps/backend/internal/office/retention/census.go new file mode 100644 index 00000000000..061f66f2561 --- /dev/null +++ b/apps/backend/internal/office/retention/census.go @@ -0,0 +1,170 @@ +package retention + +import ( + "encoding/json" + "fmt" + "sort" + "sync" + "time" +) + +// CensusState is the tri-state freshness of a thresholded table's retained +// count (AC-OFFICE-RUN-HISTORY-RETENTION-003.11): zero is a real +// measurement, so "nobody has counted yet" cannot be represented as a +// count of zero. +type CensusState int + +const ( + // CensusNotComputed is the state before the first evaluation ever + // succeeds for a table. AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and + // -004.8 both require the surface to render this honestly rather than + // as a zero count. + CensusNotComputed CensusState = iota + // CensusFresh means RetainedCount reflects the most recent evaluation, + // which succeeded. + CensusFresh + // CensusStale means the most recent evaluation failed, and the fields + // below are carried over unchanged from the last one that succeeded + // (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's "keeps the last successful + // counts, reports them stale" — extended to the unknown-status warning + // riding the same query, since neither the requirement nor the design + // says the two field groups should diverge on a failed evaluation). + CensusStale +) + +var censusStateNames = map[CensusState]string{ + CensusNotComputed: "not_computed", + CensusFresh: "fresh", + CensusStale: "stale", +} + +// MarshalJSON renders the state as its stable wire name rather than the +// underlying int, so an HTTP consumer never has to hardcode 0/1/2. +func (s CensusState) MarshalJSON() ([]byte, error) { + name, ok := censusStateNames[s] + if !ok { + return nil, fmt.Errorf("retention: unknown census state %d", s) + } + return json.Marshal(name) +} + +// UnmarshalJSON accepts only the names MarshalJSON produces. +func (s *CensusState) UnmarshalJSON(data []byte) error { + var name string + if err := json.Unmarshal(data, &name); err != nil { + return err + } + for state, candidate := range censusStateNames { + if candidate == name { + *s = state + return nil + } + } + return fmt.Errorf("retention: unknown census state %q", name) +} + +// UnknownStatusCount is one status this package does not recognize, with +// the number of retained rows currently holding it +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +type UnknownStatusCount struct { + Status string `json:"status"` + Count int64 `json:"count"` +} + +// TableCensus is one thresholded table's retained-count evaluation. +type TableCensus struct { + State CensusState `json:"state"` + RetainedCount int64 `json:"retained_count"` + AsOf time.Time `json:"as_of"` + UnknownStatuses []UnknownStatusCount `json:"unknown_statuses,omitempty"` // ascending by status; only populated for status-bearing tables + TopRoutineID string `json:"top_routine_id,omitempty"` // office_routine_runs only; empty when not applicable + TopRoutineShare float64 `json:"top_routine_share,omitempty"` // top routine's retained rows / table's retained count +} + +// RetainedCounts holds the current census result for every thresholded +// table (AC-OFFICE-RUN-HISTORY-RETENTION-003.11). +type RetainedCounts struct { + OfficeRoutineRuns TableCensus `json:"office_routine_runs"` + Runs TableCensus `json:"runs"` + RunEvents TableCensus `json:"run_events"` +} + +// summarizeStatusCensus turns a status->count breakdown into a retained +// count (the sum across every status, since "retained" is the table's +// current row count) and the status/count list, sorted ascending by +// status, for statuses belonging to neither the history nor the +// live-state set (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +func summarizeStatusCensus(counts map[string]int64, history, live []string) (retained int64, unknown []UnknownStatusCount) { + known := make(map[string]bool, len(history)+len(live)) + for _, s := range history { + known[s] = true + } + for _, s := range live { + known[s] = true + } + for status, count := range counts { + retained += count + if !known[status] { + unknown = append(unknown, UnknownStatusCount{Status: status, Count: count}) + } + } + sort.Slice(unknown, func(i, j int) bool { return unknown[i].Status < unknown[j].Status }) + return retained, unknown +} + +// CensusTracker holds the latest RetainedCounts in memory, applying the +// tri-state rule per table: a successful evaluation replaces a table's +// entry and marks it fresh; a failed evaluation leaves an already-computed +// entry in place and marks it stale, or leaves a never-computed entry as +// not-yet-computed. One table's failure never touches another table's +// entry. Safe for concurrent use: Record* is called from the scheduler +// goroutine and Snapshot from HTTP handlers. +type CensusTracker struct { + mu sync.Mutex + counts RetainedCounts +} + +// NewCensusTracker returns a tracker with every table not yet computed. +func NewCensusTracker() *CensusTracker { + return &CensusTracker{} +} + +// Snapshot returns the current RetainedCounts. +func (t *CensusTracker) Snapshot() RetainedCounts { + t.mu.Lock() + defer t.mu.Unlock() + return t.counts +} + +// RecordRoutineRuns applies an office_routine_runs evaluation outcome. +func (t *CensusTracker) RecordRoutineRuns(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.OfficeRoutineRuns = applyCensusResult(t.counts.OfficeRoutineRuns, fresh, err) +} + +// RecordRuns applies a runs evaluation outcome. +func (t *CensusTracker) RecordRuns(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.Runs = applyCensusResult(t.counts.Runs, fresh, err) +} + +// RecordRunEvents applies a run_events evaluation outcome. +func (t *CensusTracker) RecordRunEvents(fresh TableCensus, err error) { + t.mu.Lock() + defer t.mu.Unlock() + t.counts.RunEvents = applyCensusResult(t.counts.RunEvents, fresh, err) +} + +func applyCensusResult(prev, fresh TableCensus, err error) TableCensus { + if err != nil { + if prev.State == CensusNotComputed { + return prev + } + prev.State = CensusStale + return prev + } + fresh.State = CensusFresh + return fresh +} diff --git a/apps/backend/internal/office/retention/census_json_test.go b/apps/backend/internal/office/retention/census_json_test.go new file mode 100644 index 00000000000..7590ccf1ca5 --- /dev/null +++ b/apps/backend/internal/office/retention/census_json_test.go @@ -0,0 +1,101 @@ +package retention + +import ( + "encoding/json" + "testing" +) + +func TestCensusState_MarshalJSON_UsesStableNames(t *testing.T) { + cases := map[CensusState]string{ + CensusNotComputed: `"not_computed"`, + CensusFresh: `"fresh"`, + CensusStale: `"stale"`, + } + for state, want := range cases { + got, err := json.Marshal(state) + if err != nil { + t.Fatalf("Marshal(%v): %v", state, err) + } + if string(got) != want { + t.Fatalf("Marshal(%v) = %s, want %s", state, got, want) + } + } +} + +func TestCensusState_UnmarshalJSON_RoundTrips(t *testing.T) { + for _, state := range []CensusState{CensusNotComputed, CensusFresh, CensusStale} { + encoded, err := json.Marshal(state) + if err != nil { + t.Fatalf("Marshal(%v): %v", state, err) + } + var decoded CensusState + if err := json.Unmarshal(encoded, &decoded); err != nil { + t.Fatalf("Unmarshal(%s): %v", encoded, err) + } + if decoded != state { + t.Fatalf("round trip %v -> %s -> %v", state, encoded, decoded) + } + } +} + +func TestCensusState_UnmarshalJSON_RejectsUnknownName(t *testing.T) { + var state CensusState + if err := json.Unmarshal([]byte(`"bogus"`), &state); err == nil { + t.Fatal("expected an error for an unrecognized census state name") + } +} + +func TestRetainedCounts_JSONUsesSnakeCaseFieldNames(t *testing.T) { + counts := RetainedCounts{ + OfficeRoutineRuns: TableCensus{State: CensusFresh, RetainedCount: 3}, + } + encoded, err := json.Marshal(counts) + if err != nil { + t.Fatalf("Marshal: %v", err) + } + var raw map[string]json.RawMessage + if err := json.Unmarshal(encoded, &raw); err != nil { + t.Fatalf("Unmarshal into map: %v", err) + } + for _, key := range []string{"office_routine_runs", "runs", "run_events"} { + if _, ok := raw[key]; !ok { + t.Fatalf("RetainedCounts JSON missing key %q; got %s", key, encoded) + } + } +} + +func TestLastSweep_JSONUsesSnakeCaseFieldNames(t *testing.T) { + sweep := LastSweep{ + OfficeRoutineRuns: SweptTableResult{ + TableSweepResult: TableSweepResult{Deleted: 5, Backlog: true}, + Previewed: true, + WouldDelete: 7, + }, + } + encoded, err := json.Marshal(sweep) + if err != nil { + t.Fatalf("Marshal: %v", err) + } + var raw map[string]json.RawMessage + if err := json.Unmarshal(encoded, &raw); err != nil { + t.Fatalf("Unmarshal into map: %v", err) + } + for _, key := range []string{ + "started_at", "finished_at", "office_routine_runs", "runs", + "run_events", "route_attempts", "run_skills", + } { + if _, ok := raw[key]; !ok { + t.Fatalf("LastSweep JSON missing key %q; got %s", key, encoded) + } + } + + var routineRuns map[string]json.RawMessage + if err := json.Unmarshal(raw["office_routine_runs"], &routineRuns); err != nil { + t.Fatalf("Unmarshal office_routine_runs: %v", err) + } + for _, key := range []string{"deleted", "backlog", "error", "previewed", "would_delete"} { + if _, ok := routineRuns[key]; !ok { + t.Fatalf("SweptTableResult JSON missing key %q; got %s", key, raw["office_routine_runs"]) + } + } +} diff --git a/apps/backend/internal/office/retention/census_test.go b/apps/backend/internal/office/retention/census_test.go new file mode 100644 index 00000000000..c776d5bf5e3 --- /dev/null +++ b/apps/backend/internal/office/retention/census_test.go @@ -0,0 +1,133 @@ +package retention + +import ( + "errors" + "reflect" + "testing" + "time" +) + +func TestSummarizeStatusCensus_SumsAllStatusesRegardlessOfClass(t *testing.T) { + counts := map[string]int64{ + "done": 3, + "received": 2, + } + retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + if retained != 5 { + t.Fatalf("retained = %d, want 5", retained) + } + if len(unknown) != 0 { + t.Fatalf("unknown = %v, want none", unknown) + } +} + +func TestSummarizeStatusCensus_DetectsUnknownStatusesSortedAscending(t *testing.T) { + counts := map[string]int64{ + "done": 1, + "zeta": 1, + "alpha": 1, + "skipped": 1, + } + retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + if retained != 4 { + t.Fatalf("retained = %d, want 4", retained) + } + want := []UnknownStatusCount{{Status: "alpha", Count: 1}, {Status: "zeta", Count: 1}} + if !reflect.DeepEqual(unknown, want) { + t.Fatalf("unknown = %v, want %v", unknown, want) + } +} + +func TestSummarizeStatusCensus_EmptyTableIsZeroNotUnknown(t *testing.T) { + retained, unknown := summarizeStatusCensus(map[string]int64{}, RunHistoryStatuses, RunLiveStatuses) + if retained != 0 { + t.Fatalf("retained = %d, want 0", retained) + } + if unknown != nil { + t.Fatalf("unknown = %v, want nil", unknown) + } +} + +func TestCensusTracker_SnapshotStartsNotComputedForEveryTable(t *testing.T) { + tracker := NewCensusTracker() + snap := tracker.Snapshot() + for name, c := range map[string]TableCensus{ + "office_routine_runs": snap.OfficeRoutineRuns, + "runs": snap.Runs, + "run_events": snap.RunEvents, + } { + if c.State != CensusNotComputed { + t.Fatalf("%s: state = %v, want CensusNotComputed", name, c.State) + } + } +} + +func TestCensusTracker_SuccessfulEvaluationIsFresh(t *testing.T) { + tracker := NewCensusTracker() + now := time.Now().UTC() + tracker.RecordRuns(TableCensus{RetainedCount: 42, AsOf: now}, nil) + + got := tracker.Snapshot().Runs + if got.State != CensusFresh { + t.Fatalf("state = %v, want CensusFresh", got.State) + } + if got.RetainedCount != 42 { + t.Fatalf("retained = %d, want 42", got.RetainedCount) + } + if !got.AsOf.Equal(now) { + t.Fatalf("asOf = %v, want %v", got.AsOf, now) + } +} + +func TestCensusTracker_FailedEvaluationWithNoPriorSuccessStaysNotComputed(t *testing.T) { + tracker := NewCensusTracker() + tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom")) + + got := tracker.Snapshot().OfficeRoutineRuns + if got.State != CensusNotComputed { + t.Fatalf("state = %v, want CensusNotComputed", got.State) + } +} + +func TestCensusTracker_FailedEvaluationAfterSuccessKeepsLastCountsMarkedStale(t *testing.T) { + tracker := NewCensusTracker() + first := TableCensus{ + RetainedCount: 100, + AsOf: time.Now().UTC(), + UnknownStatuses: []UnknownStatusCount{{Status: "weird", Count: 1}}, + TopRoutineID: "r-1", + TopRoutineShare: 0.5, + } + tracker.RecordRoutineRuns(first, nil) + + tracker.RecordRoutineRuns(TableCensus{}, errors.New("query failed")) + + got := tracker.Snapshot().OfficeRoutineRuns + if got.State != CensusStale { + t.Fatalf("state = %v, want CensusStale", got.State) + } + if got.RetainedCount != 100 { + t.Fatalf("retained = %d, want 100 (carried over)", got.RetainedCount) + } + if !reflect.DeepEqual(got.UnknownStatuses, []UnknownStatusCount{{Status: "weird", Count: 1}}) { + t.Fatalf("unknownStatuses = %v, want carried over", got.UnknownStatuses) + } + if got.TopRoutineID != "r-1" || got.TopRoutineShare != 0.5 { + t.Fatalf("top routine attribution not carried over: %+v", got) + } +} + +func TestCensusTracker_OneTableFailureDoesNotTouchAnother(t *testing.T) { + tracker := NewCensusTracker() + tracker.RecordRuns(TableCensus{RetainedCount: 7, AsOf: time.Now().UTC()}, nil) + + tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom")) + + snap := tracker.Snapshot() + if snap.Runs.State != CensusFresh || snap.Runs.RetainedCount != 7 { + t.Fatalf("runs entry disturbed by routine_runs failure: %+v", snap.Runs) + } + if snap.OfficeRoutineRuns.State != CensusNotComputed { + t.Fatalf("routine_runs state = %v, want CensusNotComputed", snap.OfficeRoutineRuns.State) + } +} diff --git a/apps/backend/internal/office/retention/goleak_test.go b/apps/backend/internal/office/retention/goleak_test.go new file mode 100644 index 00000000000..7cfa26b30a1 --- /dev/null +++ b/apps/backend/internal/office/retention/goleak_test.go @@ -0,0 +1,15 @@ +package retention + +import ( + "testing" + + "go.uber.org/goleak" +) + +// TestMain enforces no goroutine leaks across the retention package. +// Scheduler.Start spawns a single lifecycle-managed loop goroutine; Stop +// cancels its context and waits on the WaitGroup. Regressions where Stop +// forgets to cancel or a test leaves a scheduler running surface here. +func TestMain(m *testing.M) { + goleak.VerifyTestMain(m) +} diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go new file mode 100644 index 00000000000..d68b5e83826 --- /dev/null +++ b/apps/backend/internal/office/retention/handler.go @@ -0,0 +1,141 @@ +package retention + +import ( + "errors" + "net/http" + "sync" + "time" + + "github.com/gin-gonic/gin" +) + +const responseErrorKey = "error" + +// maxRetentionSettingsBodyBytes bounds administrator-controlled JSON before +// the handler decodes it, so a malformed request cannot consume unbounded +// memory. +const maxRetentionSettingsBodyBytes = 1 << 20 + +// HandlerConfig wires the HTTP surface to the package's own stores. +type HandlerConfig struct { + SettingsStore *SettingsStore + Sweeper *Sweeper + // OnSettingsChanged, when set, is called with the normalized document + // after a successful PUT so the running scheduler re-arms its timers + // from the new settings without a restart (AC-004.5). + OnSettingsChanged func(Settings) + LogError func(string, error) +} + +// Handler serves GET/PUT /api/v1/system/retention. +type Handler struct { + config HandlerConfig + + // mu serializes a PUT's save and scheduler-apply as one critical + // section, so two concurrent PUTs cannot interleave into the scheduler + // applying the older of the two writes after the newer one is already + // stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.5's last-writer-wins). + mu sync.Mutex +} + +// NewHandler wires a Handler to its dependencies. +func NewHandler(config HandlerConfig) *Handler { + return &Handler{config: config} +} + +func (h *Handler) logError(message string, err error) { + if h.config.LogError != nil { + h.config.LogError(message, err) + } +} + +// RegisterRoutes wires GET/PUT /api/v1/system/retention: GET is +// member-readable, matching every sibling System-pages surface (storage, +// queue settings, sleep inhibition), and readable while retention is +// disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8); PUT is admin-scoped. +func RegisterRoutes(read, admin *gin.RouterGroup, handler *Handler) { + read.GET("/retention", handler.getRetention) + admin.PUT("/retention", handler.putRetention) +} + +// Status is the GET response body: effective settings, the most recently +// completed sweep (nil before the first one — AC-004.7), the +// separately-held skip record, and the per-thresholded-table retained-count +// census (AC-004.6). +type Status struct { + Settings Settings `json:"settings"` + LastSweep *LastSweep `json:"last_sweep"` + SkipCount int64 `json:"skip_count"` + LastSkipAt *time.Time `json:"last_skip_at,omitempty"` + RetainedCounts RetainedCounts `json:"retained_counts"` +} + +func (h *Handler) getRetention(c *gin.Context) { + settings, err := h.config.SettingsStore.GetSettings(c.Request.Context()) + if err != nil { + h.logError("failed to load retention settings", err) + } + + status := Status{ + Settings: settings, + RetainedCounts: h.config.Sweeper.CensusSnapshot(), + } + if last, ok := h.config.Sweeper.LastSweepSnapshot(); ok { + status.LastSweep = &last + } + if count, lastAt := h.config.Sweeper.SkipSnapshot(); count > 0 { + status.SkipCount = count + status.LastSkipAt = &lastAt + } + c.JSON(http.StatusOK, status) +} + +func (h *Handler) putRetention(c *gin.Context) { + c.Request.Body = http.MaxBytesReader(c.Writer, c.Request.Body, maxRetentionSettingsBodyBytes) + body, err := c.GetRawData() + if err != nil { + var maxBytesErr *http.MaxBytesError + if errors.As(err, &maxBytesErr) { + c.JSON(http.StatusRequestEntityTooLarge, gin.H{responseErrorKey: "request body too large"}) + return + } + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: "failed to read request body"}) + return + } + + settings, err := decodeRetentionSettings(body) + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()}) + return + } + + h.mu.Lock() + defer h.mu.Unlock() + + saved, err := h.config.SettingsStore.SaveSettings(c.Request.Context(), settings) + if err != nil { + if errors.Is(err, ErrValidation) { + c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()}) + return + } + h.logError("failed to save retention settings", err) + c.JSON(http.StatusInternalServerError, gin.H{responseErrorKey: "failed to save retention settings"}) + return + } + + if testBetweenSaveAndApply != nil { + testBetweenSaveAndApply() + } + + if h.config.OnSettingsChanged != nil { + h.config.OnSettingsChanged(saved) + } + c.JSON(http.StatusOK, saved) +} + +// testBetweenSaveAndApply, when set, runs after a PUT's SaveSettings +// commits and before OnSettingsChanged is invoked, while mu is still held — +// a deterministic seam for proving a second PUT cannot save and apply in +// between (the concurrent-PUT desync this mutex exists to prevent). Never +// set outside tests. +var testBetweenSaveAndApply func() diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go new file mode 100644 index 00000000000..a8909ef9b7b --- /dev/null +++ b/apps/backend/internal/office/retention/handler_test.go @@ -0,0 +1,442 @@ +package retention + +import ( + "bytes" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + "time" + + "github.com/gin-gonic/gin" + + "github.com/kandev/kandev/internal/auth/authn" +) + +// newTestRetentionRouter mirrors production wiring: GET is member-readable, +// PUT requires admin, matching storage's and sleep-inhibition's split +// read/admin groups. +func newTestRetentionRouter(handler *Handler) *gin.Engine { + return newTestRetentionRouterAs(handler, authn.RoleAdmin) +} + +func newTestRetentionRouterAs(handler *Handler, role authn.Role) *gin.Engine { + router := gin.New() + router.Use(func(c *gin.Context) { + authn.SetOnGin(c, authn.Identity{UserID: "user-1", Role: role}) + c.Next() + }) + read := router.Group("/api/v1/system") + admin := read.Group("", authn.RequireAdmin()) + RegisterRoutes(read, admin, handler) + return router +} + +func newTestHandler(t *testing.T) (*Handler, *Sweeper) { + t.Helper() + sweeper, _ := newTestSweeper(t) + handler := NewHandler(HandlerConfig{SettingsStore: sweeper.settingsStore, Sweeper: sweeper}) + return handler, sweeper +} + +func doRequest(router *gin.Engine, method, path string, body []byte) *httptest.ResponseRecorder { + var reader *bytes.Reader + if body != nil { + reader = bytes.NewReader(body) + } else { + reader = bytes.NewReader(nil) + } + request := httptest.NewRequest(method, path, reader) + request.Header.Set("Content-Type", "application/json") + response := httptest.NewRecorder() + router.ServeHTTP(response, request) + return response +} + +func TestGetRetention_FreshInstallReturnsDefaultsAndNilLastSweep(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var status Status + if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if status.LastSweep != nil { + t.Fatalf("LastSweep = %+v, want nil before the first sweep (AC-004.7)", status.LastSweep) + } + if status.Settings != DefaultSettings() { + t.Fatalf("Settings = %+v, want defaults", status.Settings) + } + if status.SkipCount != 0 { + t.Fatalf("SkipCount = %d, want 0", status.SkipCount) + } +} + +func TestGetRetention_NonAdminMemberCanRead(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouterAs(handler, authn.RoleMember) + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("member GET status = %d, want 200: %s", response.Code, response.Body.String()) + } +} + +func TestPutRetention_NonAdminMemberIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouterAs(handler, authn.RoleMember) + + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{}`)) + if response.Code != http.StatusForbidden { + t.Fatalf("member PUT status = %d, want 403: %s", response.Code, response.Body.String()) + } +} + +func TestGetRetention_ReflectsLastSweepAndCensus(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + ctx := t.Context() + + sweeper.RunCensus(ctx) + sweeper.RunSweep(ctx) // preview pass; still sets LastSweep + + response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var status Status + if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if status.LastSweep == nil { + t.Fatal("LastSweep = nil, want non-nil after a sweep has run") + } + if status.RetainedCounts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("RetainedCounts.OfficeRoutineRuns.State = %v, want fresh", status.RetainedCounts.OfficeRoutineRuns.State) + } +} + +func TestPutRetention_OmittedFieldTakesDocumentedDefault(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + + var saved Settings + if err := json.Unmarshal(response.Body.Bytes(), &saved); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if saved.Enabled { + t.Fatal("Enabled = true, want false (explicitly set)") + } + want := DefaultSettings() + if saved.SweepIntervalHours != want.SweepIntervalHours { + t.Fatalf("SweepIntervalHours = %d, want the default %d (omitted field)", saved.SweepIntervalHours, want.SweepIntervalHours) + } + if saved.RoutineRuns != want.RoutineRuns { + t.Fatalf("RoutineRuns = %+v, want the default %+v (omitted field)", saved.RoutineRuns, want.RoutineRuns) + } +} + +func TestPutRetention_RepeatedIdenticalWriteIsANoOp(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body, err := json.Marshal(DefaultSettings()) + if err != nil { + t.Fatalf("marshal: %v", err) + } + + first := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + second := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if first.Code != http.StatusOK || second.Code != http.StatusOK { + t.Fatalf("status = %d, %d, want 200, 200", first.Code, second.Code) + } + if first.Body.String() != second.Body.String() { + t.Fatalf("identical writes returned different documents:\n%s\n%s", first.Body.String(), second.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_ExplicitNullIsRejectedNamingTheField(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"runs": {"window_days": null}}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "runs.window_days") { + t.Fatalf("body = %s, want it to name runs.window_days", response.Body.String()) + } + + // Nothing written: a decade-long window must survive a null-rejected PUT. + settings := DefaultSettings() + settings.Runs.WindowDays = 3650 + if _, err := sweeper.settingsStore.SaveSettings(t.Context(), settings); err != nil { + t.Fatalf("seed SaveSettings: %v", err) + } + response = doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400", response.Code) + } + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored.Runs.WindowDays != 3650 { + t.Fatalf("Runs.WindowDays = %d, want 3650 unchanged (a rejected write must not destroy configured history)", stored.Runs.WindowDays) + } +} + +func TestPutRetention_UnknownFieldIsRejectedNamingTheField(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"windw_days": 30}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "windw_days") { + t.Fatalf("body = %s, want it to name the misspelled field", response.Body.String()) + } +} + +func TestPutRetention_TrailingDataAfterObjectIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": true} {"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults (nothing written on rejection)", stored) + } +} + +func TestPutRetention_TopLevelNullIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte("null")) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_OversizedBodyIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(strings.Repeat(" ", maxRetentionSettingsBodyBytes+1)) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusRequestEntityTooLarge { + t.Fatalf("status = %d, want 413: %s", response.Code, response.Body.String()) + } +} + +func TestPutRetention_FractionalNumberIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"batch_limit": 100.5}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "batch_limit") { + t.Fatalf("body = %s, want it to name batch_limit", response.Body.String()) + } +} + +func TestPutRetention_WrongTypeIsRejected(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, _ := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": "yes"}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "enabled") { + t.Fatalf("body = %s, want it to name enabled", response.Body.String()) + } +} + +func TestPutRetention_OutOfRangeIsRejectedAndLeavesStoredSettingsUnchanged(t *testing.T) { + gin.SetMode(gin.TestMode) + handler, sweeper := newTestHandler(t) + router := newTestRetentionRouter(handler) + + body := []byte(`{"batch_limit": 1}`) // below minBatchLimit=100 + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusBadRequest { + t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String()) + } + if !strings.Contains(response.Body.String(), "batch_limit") { + t.Fatalf("body = %s, want it to name batch_limit", response.Body.String()) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if stored != DefaultSettings() { + t.Fatalf("stored = %+v, want unchanged defaults", stored) + } +} + +func TestPutRetention_SuccessInvokesOnSettingsChanged(t *testing.T) { + gin.SetMode(gin.TestMode) + sweeper, _ := newTestSweeper(t) + + var got Settings + var called bool + handler := NewHandler(HandlerConfig{ + SettingsStore: sweeper.settingsStore, + Sweeper: sweeper, + OnSettingsChanged: func(s Settings) { + called = true + got = s + }, + }) + router := newTestRetentionRouter(handler) + + body := []byte(`{"enabled": false}`) + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body) + if response.Code != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String()) + } + if !called { + t.Fatal("OnSettingsChanged was not called") + } + if got.Enabled { + t.Fatal("OnSettingsChanged received Enabled=true, want false") + } +} + +// TestPutRetention_ConcurrentPUTsApplyInSaveOrder is the regression test for +// the concurrent-PUT scheduler desync found in review: without serializing +// SaveSettings and OnSettingsChanged as one critical section, a second PUT +// racing between the first's save and apply could complete its own save and +// apply entirely in between, leaving the scheduler applying the first PUT's +// now-stale settings after the second PUT's newer write already committed +// (AC-004.5's last-writer-wins). testBetweenSaveAndApply fires while the +// first PUT still holds the handler's mutex; it starts a second PUT +// concurrently and proves that second PUT cannot complete until the first +// releases the mutex, so the two applications can never interleave. +func TestPutRetention_ConcurrentPUTsApplyInSaveOrder(t *testing.T) { + gin.SetMode(gin.TestMode) + sweeper, _ := newTestSweeper(t) + + var mu sync.Mutex + var applied []bool + handler := NewHandler(HandlerConfig{ + SettingsStore: sweeper.settingsStore, + Sweeper: sweeper, + OnSettingsChanged: func(s Settings) { + mu.Lock() + applied = append(applied, s.Enabled) + mu.Unlock() + }, + }) + router := newTestRetentionRouter(handler) + + secondDone := make(chan struct{}) + secondStarted := false + + t.Cleanup(func() { testBetweenSaveAndApply = nil }) + testBetweenSaveAndApply = func() { + testBetweenSaveAndApply = nil // only race a second request once + secondStarted = true + go func() { + response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": true}`)) + if response.Code != http.StatusOK { + t.Errorf("second PUT status = %d, want 200: %s", response.Code, response.Body.String()) + } + close(secondDone) + }() + + select { + case <-secondDone: + t.Fatal("second PUT completed while the first still held the critical section") + case <-time.After(100 * time.Millisecond): + } + } + + first := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": false}`)) + if first.Code != http.StatusOK { + t.Fatalf("first PUT status = %d, want 200: %s", first.Code, first.Body.String()) + } + if !secondStarted { + t.Fatal("test hook never fired; the race was not exercised") + } + + select { + case <-secondDone: + case <-time.After(5 * time.Second): + t.Fatal("second PUT never completed after the first released the critical section") + } + + mu.Lock() + defer mu.Unlock() + if len(applied) != 2 || applied[0] != false || applied[1] != true { + t.Fatalf("OnSettingsChanged calls = %+v, want [false, true] in save order", applied) + } + + stored, err := sweeper.settingsStore.GetSettings(t.Context()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if !stored.Enabled { + t.Fatal("stored Enabled = false, want true (the second, later PUT must win)") + } +} diff --git a/apps/backend/internal/office/retention/health.go b/apps/backend/internal/office/retention/health.go new file mode 100644 index 00000000000..67fd613f5d4 --- /dev/null +++ b/apps/backend/internal/office/retention/health.go @@ -0,0 +1,240 @@ +package retention + +import ( + "context" + "fmt" + "sort" + "strings" + "time" + + "github.com/kandev/kandev/internal/health" +) + +const ( + fixURL = "/settings/system/data-storage" + fixLabel = "Review retention settings" +) + +// Checker implements health.Checker for office run history retention. Every +// issue is derived fresh from live state on each Check() — LastSweep and +// RetainedCounts are already the durable, tri-state views this design +// specifies (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7, -003.11) — so +// there is no separate stored issue map to keep in sync with them. +// +// office_retention_count_failed: is a ninth issue id beyond the +// design's closed eight-id catalogue, covering a failed census evaluation. +// It fires only on CensusStale (a table that had a successful evaluation and +// then failed); CensusNotComputed is the pre-first-success state AC-003.11 +// requires rendering as absent rather than alarming, so it raises nothing on +// its own. +// +// office_retention_threshold:
and office_retention_disabled:
+// both answer AC-003.5/-003.7's "retained count over threshold" condition, +// split by whether retention is enabled: AC-003.7 requires the disabled case +// to additionally state that retention is disabled, and the catalogue gives +// it its own id rather than a variable message under one id. +type Checker struct { + settingsStore *SettingsStore + sweeper *Sweeper + previewMarker *PreviewMarkerStore +} + +// NewChecker wires the health checker to the package's own stores. +func NewChecker(settingsStore *SettingsStore, sweeper *Sweeper, previewMarker *PreviewMarkerStore) *Checker { + return &Checker{settingsStore: settingsStore, sweeper: sweeper, previewMarker: previewMarker} +} + +func (c *Checker) Name() string { return "Office run retention" } +func (c *Checker) Category() string { return "office" } + +func (c *Checker) Check(ctx context.Context) []health.Issue { + var issues []health.Issue + + settings, err := c.settingsStore.GetSettings(ctx) + if err != nil { + issues = append(issues, issue( + "office_retention_settings_invalid", + "Retention settings unreadable", + fmt.Sprintf("Stored retention settings could not be read; using the documented defaults. (%s)", err.Error()), + )) + } + + if _, readable := c.previewMarker.Get(ctx); !readable { + issues = append(issues, issue( + "office_retention_preview_unreadable", + "Retention preview marker unreadable", + "The retention preview marker could not be read; office_routine_runs and runs will be previewed again on the next sweep rather than deleting.", + )) + } + + if lastSweep, ok := c.sweeper.LastSweepSnapshot(); ok { + issues = append(issues, sweptTableIssues(lastSweep, settings)...) + issues = append(issues, failedTableIssues(lastSweep)...) + } + + issues = append(issues, c.censusIssues(settings)...) + + sort.Slice(issues, func(i, j int) bool { return issues[i].ID < issues[j].ID }) + return issues +} + +// sweptTableIssues covers AC-003.2 (preview pending, one combined issue +// naming every swept table with a nonzero would-delete count) and AC-003.6 +// (backlog, per swept table). +func sweptTableIssues(last LastSweep, settings Settings) []health.Issue { + var issues []health.Issue + + type sweptEntry struct { + table TableName + result SweptTableResult + window int + } + entries := []sweptEntry{ + {TableOfficeRoutineRuns, last.OfficeRoutineRuns, settings.RoutineRuns.WindowDays}, + {TableRuns, last.Runs, settings.Runs.WindowDays}, + } + + var pending []string + for _, e := range entries { + if e.result.Previewed && e.result.WouldDelete > 0 { + pending = append(pending, fmt.Sprintf("%s: %d rows under a %d-day window", e.table, e.result.WouldDelete, e.window)) + } + } + if len(pending) > 0 { + issues = append(issues, issue( + "office_retention_preview_pending", + "Retention preview pending deletion", + "Deletion begins at the next scheduled sweep: "+strings.Join(pending, "; ")+".", + )) + } + + for _, e := range entries { + if !e.result.Backlog { + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_backlog:%s", e.table), + "Retention is behind", + fmt.Sprintf("%s has more eligible rows than one sweep's batch limit; %d rows were deleted this sweep and retention remains behind.", e.table, e.result.Deleted), + )) + } + return issues +} + +// failedTableIssues covers AC-002.7/AC-004.6's per-table sweep failure. Only +// office_routine_runs and runs ever carry a nonempty Err in the current +// sweep implementation — a satellite's own delete is never independently +// batched or retried — but every reported table is checked generically so a +// future failure mode on a satellite surfaces without a code change here. +func failedTableIssues(last LastSweep) []health.Issue { + entries := []struct { + table TableName + result TableSweepResult + }{ + {TableOfficeRoutineRuns, last.OfficeRoutineRuns.TableSweepResult}, + {TableRuns, last.Runs.TableSweepResult}, + {TableRunEvents, last.RunEvents}, + {"office_run_route_attempts", last.RouteAttempts}, + {"office_run_skills", last.RunSkills}, + } + var issues []health.Issue + for _, e := range entries { + if e.result.Err == "" { + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_failed:%s", e.table), + "Retention sweep failed", + fmt.Sprintf("The last sweep failed for %s: %s", e.table, e.result.Err), + )) + } + return issues +} + +// censusIssues covers AC-001.10 (unknown status), AC-003.5/-003.7 (threshold, +// split on enabled/disabled), and office_retention_count_failed. +func (c *Checker) censusIssues(settings Settings) []health.Issue { + counts := c.sweeper.CensusSnapshot() + + entries := []struct { + table TableName + census TableCensus + warnRows int + }{ + {TableOfficeRoutineRuns, counts.OfficeRoutineRuns, settings.RoutineRuns.WarnRows}, + {TableRuns, counts.Runs, settings.Runs.WarnRows}, + {TableRunEvents, counts.RunEvents, settings.RunEvents.WarnRows}, + } + + var issues []health.Issue + for _, e := range entries { + if e.census.State == CensusStale { + issues = append(issues, issue( + fmt.Sprintf("office_retention_count_failed:%s", e.table), + "Retained-row count evaluation failing", + fmt.Sprintf("%s's retained-row count could not be re-evaluated; showing the last successful count from %s.", e.table, e.census.AsOf.Format(time.RFC3339)), + )) + } + if e.census.State == CensusNotComputed { + continue + } + + if len(e.census.UnknownStatuses) > 0 { + issues = append(issues, issue( + fmt.Sprintf("office_retention_unknown_status:%s", e.table), + "Unrecognized status in retained rows", + fmt.Sprintf("%s has rows with unrecognized status values, treated as live state and never pruned: %s.", e.table, formatUnknownStatuses(e.census.UnknownStatuses)), + )) + } + + if e.warnRows <= 0 || e.census.RetainedCount <= int64(e.warnRows) { + continue + } + message := thresholdMessage(e.table, e.census, e.warnRows) + if !settings.Enabled { + issues = append(issues, issue( + fmt.Sprintf("office_retention_disabled:%s", e.table), + "Retention disabled with rows over threshold", + message+" Retention is disabled, so this table is not being pruned.", + )) + continue + } + issues = append(issues, issue( + fmt.Sprintf("office_retention_threshold:%s", e.table), + "Retained rows over threshold", + message, + )) + } + return issues +} + +// formatUnknownStatuses renders each unrecognized status with its row +// count, in the ascending order summarizeStatusCensus already sorted them +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +func formatUnknownStatuses(unknown []UnknownStatusCount) string { + parts := make([]string, len(unknown)) + for i, u := range unknown { + parts[i] = fmt.Sprintf("%s (%d)", u.Status, u.Count) + } + return strings.Join(parts, ", ") +} + +func thresholdMessage(table TableName, census TableCensus, warnRows int) string { + message := fmt.Sprintf("%s has %d retained rows, over its threshold of %d.", table, census.RetainedCount, warnRows) + if census.TopRoutineID != "" { + message += fmt.Sprintf(" Routine %s holds the largest share of retained rows, %.1f%%.", census.TopRoutineID, census.TopRoutineShare*100) + } + return message +} + +func issue(id, title, message string) health.Issue { + return health.Issue{ + ID: id, + Category: "office", + Title: title, + Message: message, + Severity: health.SeverityWarning, + FixURL: fixURL, + FixLabel: fixLabel, + } +} diff --git a/apps/backend/internal/office/retention/health_test.go b/apps/backend/internal/office/retention/health_test.go new file mode 100644 index 00000000000..a768495a26b --- /dev/null +++ b/apps/backend/internal/office/retention/health_test.go @@ -0,0 +1,381 @@ +package retention + +import ( + "context" + "errors" + "strings" + "testing" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/health" +) + +var errCensusEvaluation = errors.New("census evaluation failed") + +func newTestChecker(t *testing.T) (*Checker, *Sweeper, *sqlx.DB) { + t.Helper() + sweeper, conn := newTestSweeper(t) + checker := NewChecker(sweeper.settingsStore, sweeper, sweeper.previewMarker) + return checker, sweeper, conn +} + +func issueIDs(issues []health.Issue) []string { + ids := make([]string, 0, len(issues)) + for _, i := range issues { + ids = append(ids, i.ID) + } + return ids +} + +func hasIssue(issues []health.Issue, id string) bool { + for _, i := range issues { + if i.ID == id { + return true + } + } + return false +} + +func issueMessage(t *testing.T, issues []health.Issue, id string) string { + t.Helper() + for _, i := range issues { + if i.ID == id { + return i.Message + } + } + t.Fatalf("no issue with id %q in %v", id, issueIDs(issues)) + return "" +} + +func TestChecker_NameAndCategory(t *testing.T) { + checker, _, _ := newTestChecker(t) + if checker.Name() != "Office run retention" { + t.Fatalf("Name() = %q", checker.Name()) + } + if checker.Category() != "office" { + t.Fatalf("Category() = %q", checker.Category()) + } +} + +func TestChecker_FreshInstallNoSweepNoIssues(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + sweeper.RunCensus(ctx) // AC-003.11: counts available before any sweep + + issues := checker.Check(ctx) + if len(issues) != 0 { + t.Fatalf("issues = %v, want none on a fresh, empty, enabled install", issueIDs(issues)) + } +} + +func TestChecker_SettingsInvalidRaisesGlobalIssue(t *testing.T) { + checker, _, conn := newTestChecker(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_settings_invalid") { + t.Fatalf("issues = %v, want office_retention_settings_invalid", issueIDs(issues)) + } +} + +func TestChecker_PreviewMarkerUnreadableRaisesGlobalIssue(t *testing.T) { + checker, _, conn := newTestChecker(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention_preview_completed', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable preview marker: %v", err) + } + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_preview_unreadable") { + t.Fatalf("issues = %v, want office_retention_preview_unreadable", issueIDs(issues)) + } +} + +func TestChecker_PreviewPendingNamesTablesWithNonzeroWouldDelete(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) // preview pass, both tables have 1 eligible row + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_preview_pending") { + t.Fatalf("issues = %v, want office_retention_preview_pending", issueIDs(issues)) + } +} + +func TestChecker_PreviewWithNothingEligibleRaisesNoWarning(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + sweeper.RunSweep(ctx) // preview pass, nothing seeded, both tables report zero + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_preview_pending") { + t.Fatalf("issues = %v, want no preview_pending when the preview found nothing (AC-003.3)", issueIDs(issues)) + } +} + +func TestChecker_BacklogRaisesPerSweptTable(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass, 105 eligible > batch limit 100 + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_backlog:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_backlog:office_routine_runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_backlog:runs") { + t.Fatalf("issues = %v, want no backlog issue for runs (never seeded)", issueIDs(issues)) + } +} + +func TestChecker_PreviewedTableNeverRaisesBacklog(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass only: 105 eligible, never backlog (AC-002.3) + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_backlog:office_routine_runs") { + t.Fatalf("issues = %v, want no backlog issue for a preview pass regardless of eligible count", issueIDs(issues)) + } +} + +func TestChecker_AbandonedRunsBatchRaisesFailedIssueForRunsOnly(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + old := daysAgo(60) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + + testBeforeSelectEligibleRunIDs = func(attempt int) { + if attempt == 0 { + return + } + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID) + } + testAfterSelectEligibleRunIDs = func(int, []string) { + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID) + } + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + + sweeper.RunSweep(ctx) // deleting pass: forced to abandon + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_failed:runs") { + t.Fatalf("issues = %v, want office_retention_failed:runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_failed:run_events") { + t.Fatalf("issues = %v, want no failed issue for run_events (the rollback restored it, not a failure of its own)", issueIDs(issues)) + } +} + +func TestChecker_UnknownStatusRaisesWhileDisabledAndBeforeAnySweep(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2)) + + sweeper.RunCensus(ctx) // AC-003.11: census runs independent of the sweep/enabled state + + issues := checker.Check(ctx) + const id = "office_retention_unknown_status:office_routine_runs" + if !hasIssue(issues, id) { + t.Fatalf("issues = %v, want %s (001.10, disabled, no sweep ever ran)", issueIDs(issues), id) + } + if message := issueMessage(t, issues, id); !strings.Contains(message, "quarantined (2)") { + t.Fatalf("message = %q, want it to name the unrecognized status with its row count", message) + } +} + +func TestChecker_ThresholdExceededWhileEnabledRaisesThresholdID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_threshold:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_threshold:office_routine_runs", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want no disabled-variant issue while enabled", issueIDs(issues)) + } +} + +func TestChecker_ThresholdExceededWhileDisabledRaisesDisabledID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + settings.RoutineRuns.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_disabled:office_routine_runs (AC-003.7)", issueIDs(issues)) + } + if hasIssue(issues, "office_retention_threshold:office_routine_runs") { + t.Fatalf("issues = %v, want no plain threshold issue while disabled", issueIDs(issues)) + } +} + +func TestChecker_ZeroWarnRowsDisablesThresholdIssue(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + seedRoutine(t, conn, "r-1") + for i := 0; i < 5; i++ { + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(i+1)), daysAgo(i+1)) + } + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_threshold:office_routine_runs") || hasIssue(issues, "office_retention_disabled:office_routine_runs") { + t.Fatalf("issues = %v, want no threshold issue when warn_rows=0", issueIDs(issues)) + } +} + +func TestChecker_CountFailedOnlyAfterAPriorSuccess(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + sweeper.RunCensus(ctx) // succeeds once + + sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation) + + issues := checker.Check(ctx) + if !hasIssue(issues, "office_retention_count_failed:office_routine_runs") { + t.Fatalf("issues = %v, want office_retention_count_failed:office_routine_runs after a prior success then a failure", issueIDs(issues)) + } +} + +func TestChecker_NotComputedNeverRaisesCountFailed(t *testing.T) { + checker, sweeper, _ := newTestChecker(t) + ctx := context.Background() + + sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation) // fails, no prior success + + issues := checker.Check(ctx) + if hasIssue(issues, "office_retention_count_failed:office_routine_runs") { + t.Fatalf("issues = %v, want no count_failed issue before any evaluation ever succeeded (AC-003.11's not-yet-computed state)", issueIDs(issues)) + } +} + +func TestChecker_IssuesSortedByID(t *testing.T) { + checker, sweeper, conn := newTestChecker(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.RoutineRuns.WarnRows = 1 + settings.Runs.WarnRows = 1 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2)) + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(2)), daysAgo(2)) + sweeper.RunCensus(ctx) + + issues := checker.Check(ctx) + ids := issueIDs(issues) + for i := 1; i < len(ids); i++ { + if ids[i-1] > ids[i] { + t.Fatalf("issues not sorted by id: %v", ids) + } + } +} diff --git a/apps/backend/internal/office/retention/lock.go b/apps/backend/internal/office/retention/lock.go new file mode 100644 index 00000000000..a96802c3e56 --- /dev/null +++ b/apps/backend/internal/office/retention/lock.go @@ -0,0 +1,107 @@ +package retention + +import ( + "context" + "database/sql/driver" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" +) + +// advisoryLockKey is retention's own PostgreSQL advisory lock namespace, +// distinct from every other hashtextextended(?, 0) call site in the repo +// (participants.go, secrets/sqlite_store.go, workflow/repository/phase2_sqlite.go) +// so a sweep never contends with participant-seat, secret-transfer, or +// workflow-phase locking. +const advisoryLockKey = "office_run_retention_sweep" + +const unlockTimeout = 5 * time.Second + +// sweepSession is a PostgreSQL session-scoped, non-blocking exclusivity +// lock for one sweep (AC-OFFICE-RUN-HISTORY-RETENTION-002.12): +// +// - Single connection budget: every statement the sweep issues — count, +// delete, census — runs through this one dedicated connection, +// returned by queryer(), rather than reserving it for the lock alone +// and running batches on the shared pool. Reserving a second +// connection is exactly what a maxOpenConns=1 pool (what +// testutil.OpenIsolatedPostgres, the mandated Postgres-gated test +// harness, sets) cannot supply; one connection total removes the +// deadlock. +// - Exclusivity window: because every sweep statement runs on the lock +// connection, there is no window where work proceeds on a different +// connection after the session died — if the session ends, the very +// next statement on it fails immediately instead of continuing to run +// against the pool while another backend has already re-acquired the +// lock. The between-tables alive() check the design specifies is kept +// anyway, as a cheap early exit before starting a table's work rather +// than the only guard against loss of exclusivity. +// - Release safety: release() unlocks and closes on an independent +// context, not the sweep's (which may already be cancelled), with a +// bounded timeout, and discards the connection via driver.ErrBadConn +// whenever the unlock did not provably succeed — so database/sql +// never pools a session that may still hold the lock, which is what +// the design's "cannot wedge retention permanently" claim actually +// requires (internal/db sets no ConnMaxLifetime). +type sweepSession struct { + conn *sqlx.Conn +} + +// acquireSweepSession tries to take the advisory lock on a fresh dedicated +// connection. ok is false when another backend already holds it or the +// connection could not be checked out; the caller records a skip and +// returns rather than retrying (AC-OFFICE-RUN-HISTORY-RETENTION-002.12). +func acquireSweepSession(ctx context.Context, pool *db.Pool) (*sweepSession, bool, error) { + conn, err := pool.Writer().Connx(ctx) + if err != nil { + return nil, false, err + } + + var acquired bool + err = conn.GetContext(ctx, &acquired, `SELECT pg_try_advisory_lock(hashtextextended($1, 0))`, advisoryLockKey) + if err != nil { + _ = conn.Close() + return nil, false, err + } + if !acquired { + _ = conn.Close() + return nil, false, nil + } + return &sweepSession{conn: conn}, true, nil +} + +// queryer is the connection every sweep statement must run through for the +// whole sweep's duration (see the single-connection-budget and +// exclusivity-window invariants on sweepSession above). +func (s *sweepSession) queryer() queryer { + return s.conn +} + +// alive reports whether the lock connection is still usable. Checked +// between tables; the sweep stops before the next table when this returns +// false and does not attempt to re-acquire, since a re-acquisition after +// another backend has taken the lock would produce exactly the concurrent +// sweep AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent. +func (s *sweepSession) alive(ctx context.Context) bool { + return s.conn.PingContext(ctx) == nil +} + +// release unlocks and closes the session. Always safe to call once; never +// call it twice. +func (s *sweepSession) release() { + ctx, cancel := context.WithTimeout(context.Background(), unlockTimeout) + defer cancel() + + var unlocked bool + err := s.conn.GetContext(ctx, &unlocked, `SELECT pg_advisory_unlock(hashtextextended($1, 0))`, advisoryLockKey) + if err != nil || !unlocked { + // The unlock did not provably succeed: force database/sql to + // discard this connection instead of returning it to the pool, + // so a session that may still hold the lock can never be reused + // by a later, unrelated caller. + _ = s.conn.Raw(func(driverConn any) error { return driver.ErrBadConn }) + } + _ = s.conn.Close() +} diff --git a/apps/backend/internal/office/retention/lock_postgres_test.go b/apps/backend/internal/office/retention/lock_postgres_test.go new file mode 100644 index 00000000000..8912887e0aa --- /dev/null +++ b/apps/backend/internal/office/retention/lock_postgres_test.go @@ -0,0 +1,141 @@ +package retention + +import ( + "context" + "testing" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/testutil" +) + +// TestSweepSession_SecondBackendSkipsRatherThanBlocks proves +// AC-OFFICE-RUN-HISTORY-RETENTION-002.12: a second backend racing for the +// same advisory lock gets ok=false immediately rather than waiting. +func TestSweepSession_SecondBackendSkipsRatherThanBlocks(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + first := testutil.OpenIsolatedPostgres(t, dsn) + firstPool := db.NewPool(first, first) + winner, ok, err := acquireSweepSession(ctx, firstPool) + if err != nil { + t.Fatalf("acquireSweepSession (winner): %v", err) + } + if !ok { + t.Fatal("winner: ok = false, want true") + } + defer winner.release() + + second := testutil.OpenIsolatedPostgres(t, dsn) + secondPool := db.NewPool(second, second) + loser, ok, err := acquireSweepSession(ctx, secondPool) + if err != nil { + t.Fatalf("acquireSweepSession (loser): %v", err) + } + if ok { + loser.release() + t.Fatal("loser: ok = true, want false") + } +} + +// TestSweepSession_RunsOnMaxOpenConnsOnePool is F27's regression test: the +// mandated Postgres test harness (testutil.OpenIsolatedPostgres) opens with +// SetMaxOpenConns(1). Reserving the lock connection AND running batches on +// the shared pool would deadlock there, since no second connection is ever +// available. Routing every statement through the lock connection's own +// queryer() must not deadlock and must give a winner more than one table's +// worth of work to do, so a transaction-scoped lock would have incorrectly +// looked sufficient here. +func TestSweepSession_RunsOnMaxOpenConnsOnePool(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + conn := testutil.OpenIsolatedPostgres(t, dsn) + pool := db.NewPool(conn, conn) + + session, ok, err := acquireSweepSession(ctx, pool) + if err != nil { + t.Fatalf("acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("ok = false, want true") + } + defer session.release() + + q := session.queryer() + for i := 0; i < 3; i++ { + var one int + if err := q.GetContext(ctx, &one, `SELECT 1`); err != nil { + t.Fatalf("query %d on lock connection: %v", i, err) + } + if one != 1 { + t.Fatalf("query %d = %d, want 1", i, one) + } + } +} + +// TestSweepSession_ReleaseAllowsReacquisition proves release() actually +// drops the lock rather than merely closing a connection database/sql +// might still consider live for pooling purposes. +func TestSweepSession_ReleaseAllowsReacquisition(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + conn := testutil.OpenIsolatedPostgres(t, dsn) + pool := db.NewPool(conn, conn) + + first, ok, err := acquireSweepSession(ctx, pool) + if err != nil || !ok { + t.Fatalf("first acquire: ok=%v err=%v", ok, err) + } + first.release() + + second, ok, err := acquireSweepSession(ctx, pool) + if err != nil { + t.Fatalf("second acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("second acquire: ok = false, want true after release") + } + second.release() +} + +// TestSweepSession_AliveFalseAfterSessionTerminated proves the +// between-tables liveness check (F25's early exit) actually detects a +// dropped session, and that PostgreSQL's own advisory-lock self-healing +// (F26's justification for a session lock over a lease row) lets a new +// session acquire afterward. +func TestSweepSession_AliveFalseAfterSessionTerminated(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + admin := testutil.OpenIsolatedPostgres(t, dsn) + adminPool := db.NewPool(admin, admin) + + victimConn := testutil.OpenIsolatedPostgres(t, dsn) + victimPool := db.NewPool(victimConn, victimConn) + victim, ok, err := acquireSweepSession(ctx, victimPool) + if err != nil || !ok { + t.Fatalf("victim acquire: ok=%v err=%v", ok, err) + } + + var pid int + if err := victim.conn.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil { + t.Fatalf("select pg_backend_pid: %v", err) + } + if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil { + t.Fatalf("terminate victim backend: %v", err) + } + + if victim.alive(ctx) { + t.Fatal("alive() = true after backend termination, want false") + } + + replacement, ok, err := acquireSweepSession(ctx, adminPool) + if err != nil { + t.Fatalf("replacement acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("replacement: ok = false, want true (terminated session must release the lock)") + } + replacement.release() +} diff --git a/apps/backend/internal/office/retention/metrics_vars.go b/apps/backend/internal/office/retention/metrics_vars.go new file mode 100644 index 00000000000..dc12c9cd591 --- /dev/null +++ b/apps/backend/internal/office/retention/metrics_vars.go @@ -0,0 +1,55 @@ +package retention + +import ( + "expvar" + "strings" +) + +// expvar maps published at package init, exposed via stdlib's /debug/vars +// handler, mirroring internal/office/scheduler/metrics_vars.go's label +// model. Development convenience only (see the design's expvar +// section) — nothing in REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on +// these; the durable operator surfaces are structured logs and +// health.Issue. +var ( + retentionSweepTotal = expvar.NewMap("office_retention_sweep_total") + retentionDeletedTotal = expvar.NewMap("office_retention_deleted_total") + retentionCensusTotal = expvar.NewMap("office_retention_census_total") +) + +// metricLabel builds a "k1=v1;k2=v2;..." label string for an expvar map +// key, matching office/scheduler's format so a downstream parser handles +// both packages identically. +func metricLabel(pairs ...string) string { + if len(pairs)%2 != 0 { + return "" + } + parts := make([]string, 0, len(pairs)/2) + for i := 0; i < len(pairs); i += 2 { + parts = append(parts, pairs[i]+"="+pairs[i+1]) + } + return strings.Join(parts, ";") +} + +func incSweepCompleted() { + retentionSweepTotal.Add(metricLabel("outcome", "completed"), 1) +} + +func incSweepSkipped() { + retentionSweepTotal.Add(metricLabel("outcome", "skipped"), 1) +} + +func incDeleted(table TableName, n int64) { + if n <= 0 { + return + } + retentionDeletedTotal.Add(metricLabel("table", string(table)), n) +} + +func incCensus(table TableName, err error) { + outcome := "fresh" + if err != nil { + outcome = "stale" + } + retentionCensusTotal.Add(metricLabel("table", string(table), "outcome", outcome), 1) +} diff --git a/apps/backend/internal/office/retention/metrics_vars_test.go b/apps/backend/internal/office/retention/metrics_vars_test.go new file mode 100644 index 00000000000..d9b25a1856d --- /dev/null +++ b/apps/backend/internal/office/retention/metrics_vars_test.go @@ -0,0 +1,116 @@ +package retention + +import ( + "errors" + "expvar" + "strconv" + "strings" + "testing" +) + +// readCounter walks the expvar map looking for a key that matches the +// supplied prefix. Returns 0 when no key matches. The prefix match keeps +// the assertion robust against process-wide test pollution. +func readCounter(t *testing.T, m *expvar.Map, prefix string) int64 { + t.Helper() + var total int64 + m.Do(func(kv expvar.KeyValue) { + if !strings.HasPrefix(kv.Key, prefix) { + return + } + n, err := strconv.ParseInt(kv.Value.String(), 10, 64) + if err != nil { + t.Fatalf("counter %q value not int: %s", kv.Key, kv.Value.String()) + } + total += n + }) + return total +} + +func TestMetricLabel(t *testing.T) { + cases := []struct { + name string + pairs []string + want string + }{ + {"single_pair", []string{"table", "runs"}, "table=runs"}, + {"odd_args_returns_empty", []string{"table"}, ""}, + {"two_pairs", []string{"table", "runs", "outcome", "completed"}, "table=runs;outcome=completed"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := metricLabel(tc.pairs...); got != tc.want { + t.Errorf("metricLabel(%v) = %q, want %q", tc.pairs, got, tc.want) + } + }) + } +} + +func TestIncSweepCompletedAndSkipped(t *testing.T) { + beforeCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed")) + incSweepCompleted() + afterCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed")) + if afterCompleted-beforeCompleted != 1 { + t.Errorf("sweep completed delta = %d, want 1", afterCompleted-beforeCompleted) + } + + beforeSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped")) + incSweepSkipped() + afterSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped")) + if afterSkipped-beforeSkipped != 1 { + t.Errorf("sweep skipped delta = %d, want 1", afterSkipped-beforeSkipped) + } +} + +func TestIncDeleted_ZeroIsNotRecorded(t *testing.T) { + label := metricLabel("table", "test_table_zero") + before := readCounter(t, retentionDeletedTotal, label) + incDeleted("test_table_zero", 0) + after := readCounter(t, retentionDeletedTotal, label) + if after != before { + t.Errorf("delta = %d, want 0 (a zero delete must not create a counter entry)", after-before) + } +} + +func TestIncDeleted_PositiveIsRecorded(t *testing.T) { + label := metricLabel("table", "test_table_positive") + before := readCounter(t, retentionDeletedTotal, label) + incDeleted("test_table_positive", 7) + after := readCounter(t, retentionDeletedTotal, label) + if after-before != 7 { + t.Errorf("delta = %d, want 7", after-before) + } +} + +func TestIncCensus_NilErrorIsFresh(t *testing.T) { + label := metricLabel("table", "test_census_fresh", "outcome", "fresh") + before := readCounter(t, retentionCensusTotal, label) + incCensus("test_census_fresh", nil) + after := readCounter(t, retentionCensusTotal, label) + if after-before != 1 { + t.Errorf("fresh delta = %d, want 1", after-before) + } +} + +func TestIncCensus_ErrorIsStale(t *testing.T) { + label := metricLabel("table", "test_census_stale", "outcome", "stale") + before := readCounter(t, retentionCensusTotal, label) + incCensus("test_census_stale", errors.New("boom")) + after := readCounter(t, retentionCensusTotal, label) + if after-before != 1 { + t.Errorf("stale delta = %d, want 1", after-before) + } +} + +func TestExpvarMapsPublishedAtKnownNames(t *testing.T) { + expected := []string{ + "office_retention_sweep_total", + "office_retention_deleted_total", + "office_retention_census_total", + } + for _, name := range expected { + if expvar.Get(name) == nil { + t.Errorf("expvar %q not published — /debug/vars consumers will miss it", name) + } + } +} diff --git a/apps/backend/internal/office/retention/policy.go b/apps/backend/internal/office/retention/policy.go new file mode 100644 index 00000000000..b3b7b4b6370 --- /dev/null +++ b/apps/backend/internal/office/retention/policy.go @@ -0,0 +1,59 @@ +package retention + +// StatusClass distinguishes a row eligible for age-based deletion from a row +// a live decision still reads, and from a row whose status this package does +// not recognize at all — a status classification failure, not a bare bool, +// so an unrecognized status can be told apart from a recognized live-state +// one (AC-OFFICE-RUN-HISTORY-RETENTION-001.10). +type StatusClass int + +const ( + // StatusUnknown is a status belonging to neither a table's history set + // nor its live-state set. Treated as live state and warned about. + StatusUnknown StatusClass = iota + StatusHistory + StatusLiveState +) + +// RoutineRunHistoryStatuses are the office_routine_runs statuses eligible +// for age-based deletion (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). +var RoutineRunHistoryStatuses = []string{"skipped", "coalesced", "failed", "done", "cancelled"} + +// RoutineRunLiveStatuses are the office_routine_runs statuses that are +// never age-pruned, at any age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). +var RoutineRunLiveStatuses = []string{"received", "task_created"} + +// RunHistoryStatuses are the runs statuses eligible for age-based deletion +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). cancelled is history: its only +// writer, CancelRunsWhere, moves a row there from queued/claimed and stamps +// finished_at in the same statement. +var RunHistoryStatuses = []string{"finished", "failed", "cancelled"} + +// RunLiveStatuses are the runs statuses that are never age-pruned, at any +// age, including a run parked for a future routing retry +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). +var RunLiveStatuses = []string{"queued", "claimed"} + +// ClassifyRoutineRunStatus classifies an office_routine_runs.status value. +func ClassifyRoutineRunStatus(status string) StatusClass { + return classify(status, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) +} + +// ClassifyRunStatus classifies a runs.status value. +func ClassifyRunStatus(status string) StatusClass { + return classify(status, RunHistoryStatuses, RunLiveStatuses) +} + +func classify(status string, history, live []string) StatusClass { + for _, s := range history { + if s == status { + return StatusHistory + } + } + for _, s := range live { + if s == status { + return StatusLiveState + } + } + return StatusUnknown +} diff --git a/apps/backend/internal/office/retention/policy_test.go b/apps/backend/internal/office/retention/policy_test.go new file mode 100644 index 00000000000..d4626a1fc20 --- /dev/null +++ b/apps/backend/internal/office/retention/policy_test.go @@ -0,0 +1,98 @@ +package retention + +import ( + "sort" + "testing" + + "github.com/kandev/kandev/internal/office/models" +) + +func TestClassifyRoutineRunStatus(t *testing.T) { + for _, s := range RoutineRunHistoryStatuses { + if got := ClassifyRoutineRunStatus(s); got != StatusHistory { + t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusHistory", s, got) + } + } + for _, s := range RoutineRunLiveStatuses { + if got := ClassifyRoutineRunStatus(s); got != StatusLiveState { + t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusLiveState", s, got) + } + } + if got := ClassifyRoutineRunStatus("some_future_status"); got != StatusUnknown { + t.Errorf("ClassifyRoutineRunStatus(unrecognized) = %v, want StatusUnknown", got) + } +} + +func TestClassifyRunStatus(t *testing.T) { + for _, s := range RunHistoryStatuses { + if got := ClassifyRunStatus(s); got != StatusHistory { + t.Errorf("ClassifyRunStatus(%q) = %v, want StatusHistory", s, got) + } + } + for _, s := range RunLiveStatuses { + if got := ClassifyRunStatus(s); got != StatusLiveState { + t.Errorf("ClassifyRunStatus(%q) = %v, want StatusLiveState", s, got) + } + } + if got := ClassifyRunStatus("some_future_status"); got != StatusUnknown { + t.Errorf("ClassifyRunStatus(unrecognized) = %v, want StatusUnknown", got) + } +} + +// TestRoutineRunStatusSets_CoverEveryEnumValue pins the closed-set claim in +// the system design: history + live-state must equal exactly the +// office/models RoutineRunStatus enumeration. A future eighth status added +// to enums.go without updating this file fails here rather than silently +// falling through ClassifyRoutineRunStatus's StatusUnknown branch on every +// install that never runs this test — the exact gap that hid `cancelled`. +func TestRoutineRunStatusSets_CoverEveryEnumValue(t *testing.T) { + wantAll := []string{ + models.RoutineRunStatusReceived.String(), + models.RoutineRunStatusTaskCreated.String(), + models.RoutineRunStatusSkipped.String(), + models.RoutineRunStatusCoalesced.String(), + models.RoutineRunStatusFailed.String(), + models.RoutineRunStatusDone.String(), + models.RoutineRunStatusCancelled.String(), + } + gotAll := append(append([]string{}, RoutineRunHistoryStatuses...), RoutineRunLiveStatuses...) + assertSameSet(t, "office_routine_runs", wantAll, gotAll) +} + +// TestRunStatusSets_CoverEveryLiteralSQLWriter pins the closed-set claim for +// runs.status. models.RunStatus itself is incomplete (it does not list +// "cancelled", even though CancelRunsWhere in +// internal/runs/repository/sqlite/cancel.go writes it) — this test asserts +// against the actual literal statuses written by SQL in that package +// instead, per the system design's Testing section, so the enum's own +// incompleteness cannot mask a gap here the way it masked `cancelled` +// before this capability existed. +func TestRunStatusSets_CoverEveryLiteralSQLWriter(t *testing.T) { + // Literal statuses written by internal/runs/repository/sqlite: + // cancel.go: 'cancelled' (CancelRunsWhere) + // runs.go:172: 'claimed' (ClaimNextRun family) + // runs.go:512: 'claimed' (claim by reason) + // runs.go:539: 'queued' (ScheduleRetry) + // runs.go:562: 'queued' (recoverStaleClaimed) + // runs.go:195/737: status passed as a bound parameter, produced by + // callers with 'finished' or 'failed' (FinishRun/MarkRunFailed). + wantAll := []string{"queued", "claimed", "finished", "failed", "cancelled"} + gotAll := append(append([]string{}, RunHistoryStatuses...), RunLiveStatuses...) + assertSameSet(t, "runs", wantAll, gotAll) +} + +func assertSameSet(t *testing.T, table string, want, got []string) { + t.Helper() + w := append([]string{}, want...) + g := append([]string{}, got...) + sort.Strings(w) + sort.Strings(g) + if len(w) != len(g) { + t.Fatalf("%s: status set size = %d, want %d (got=%v want=%v)", table, len(g), len(w), g, w) + } + for i := range w { + if w[i] != g[i] { + t.Fatalf("%s: status set mismatch at %d: got %v, want %v", table, i, g, w) + } + } +} diff --git a/apps/backend/internal/office/retention/preview_marker.go b/apps/backend/internal/office/retention/preview_marker.go new file mode 100644 index 00000000000..ef8942298f4 --- /dev/null +++ b/apps/backend/internal/office/retention/preview_marker.go @@ -0,0 +1,119 @@ +package retention + +import ( + "context" + "database/sql" + "encoding/json" + "errors" + "time" + + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// previewMarkerKey is the settings-store key holding the per-swept-table +// preview completion marker, a JSON object keyed by table name. +const previewMarkerKey = "office_run_retention_preview_completed" + +// PreviewMarker records, per swept table, the timestamp its preview +// completed. A table absent from the map has never completed a preview. +type PreviewMarker map[TableName]time.Time + +// PreviewMarkerStore persists and reads the preview marker. +type PreviewMarkerStore struct { + store *systemsettings.Store +} + +// NewPreviewMarkerStore wraps the shared key/value settings store. +func NewPreviewMarkerStore(store *systemsettings.Store) *PreviewMarkerStore { + return &PreviewMarkerStore{store: store} +} + +// Get reads the marker. readable is false only when a document is present +// but cannot be read or parsed (AC-OFFICE-RUN-HISTORY-RETENTION-003.10); a +// document that was never written is the ordinary fresh-install state and +// reports readable=true with an empty marker, not an error. Either way, a +// table absent from the returned marker has not completed a preview. +func (s *PreviewMarkerStore) Get(ctx context.Context) (PreviewMarker, bool) { + raw, found, err := s.store.Get(ctx, previewMarkerKey) + if err != nil { + return PreviewMarker{}, false + } + if !found { + return PreviewMarker{}, true + } + var doc PreviewMarker + if err := json.Unmarshal(raw, &doc); err != nil { + return PreviewMarker{}, false + } + if doc == nil { + doc = PreviewMarker{} + } + return doc, true +} + +// MarkCompleted records that table's preview as completed at the given +// time. It is only ever called after that table's preview evaluation +// completed successfully, is never cleared by a settings change or +// restart, and never touches another table's entry. If the stored document +// was corrupt, this write replaces it with a fresh document carrying only +// this table's entry — the safe direction, since a spurious re-preview of +// another table costs one sweep and deletes nothing. +func (s *PreviewMarkerStore) MarkCompleted(ctx context.Context, table TableName, at time.Time) error { + marker, readable := s.Get(ctx) + if !readable { + marker = PreviewMarker{} + } + marker[table] = at + raw, err := json.Marshal(marker) + if err != nil { + return err + } + return s.store.Save(ctx, previewMarkerKey, raw) +} + +// GetWith is Get against an explicit connection instead of the shared +// settings pool. Required mid-sweep on PostgreSQL: every statement in a +// sweep must run on the session holding the advisory lock (see sweep.go +// and lock.go), and going through the pool here would request a second +// connection, which deadlocks under a maxOpenConns=1 pool (the mandated +// Postgres-gated test harness). This bypasses systemsettings.Store and +// reads the key/value pair directly, so it depends on that package's +// `settings` table keeping its `key`/`value` column names. +func (s *PreviewMarkerStore) GetWith(ctx context.Context, q queryer) (PreviewMarker, bool) { + var raw string + err := q.GetContext(ctx, &raw, q.Rebind(`SELECT value FROM settings WHERE key = ?`), previewMarkerKey) + if err != nil { + if errors.Is(err, sql.ErrNoRows) { + return PreviewMarker{}, true + } + return PreviewMarker{}, false + } + var doc PreviewMarker + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return PreviewMarker{}, false + } + if doc == nil { + doc = PreviewMarker{} + } + return doc, true +} + +// MarkCompletedWith is MarkCompleted against an explicit connection; see +// GetWith. +func (s *PreviewMarkerStore) MarkCompletedWith(ctx context.Context, q queryer, table TableName, at time.Time) error { + marker, readable := s.GetWith(ctx, q) + if !readable { + marker = PreviewMarker{} + } + marker[table] = at + raw, err := json.Marshal(marker) + if err != nil { + return err + } + _, err = q.ExecContext(ctx, q.Rebind(` + INSERT INTO settings (key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at + `), previewMarkerKey, string(raw), time.Now().UTC()) + return err +} diff --git a/apps/backend/internal/office/retention/preview_marker_test.go b/apps/backend/internal/office/retention/preview_marker_test.go new file mode 100644 index 00000000000..8080eb380dc --- /dev/null +++ b/apps/backend/internal/office/retention/preview_marker_test.go @@ -0,0 +1,117 @@ +package retention + +import ( + "context" + "testing" + "time" +) + +func TestPreviewMarkerStore_MissingIsReadableAndEmpty(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + + marker, readable := store.Get(context.Background()) + if !readable { + t.Fatalf("Get() readable = false, want true for a never-written marker") + } + if len(marker) != 0 { + t.Fatalf("Get() marker = %+v, want empty", marker) + } +} + +func TestPreviewMarkerStore_MarkCompletedThenGetRoundTrips(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + ctx := context.Background() + + at := time.Date(2026, 9, 9, 12, 0, 0, 0, time.UTC) + if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, at); err != nil { + t.Fatalf("MarkCompleted: %v", err) + } + + marker, readable := store.Get(ctx) + if !readable { + t.Fatalf("Get() readable = false after a valid write") + } + got, ok := marker[TableOfficeRoutineRuns] + if !ok { + t.Fatalf("marker missing office_routine_runs entry: %+v", marker) + } + if !got.Equal(at) { + t.Fatalf("marker[office_routine_runs] = %v, want %v", got, at) + } + if _, ok := marker[TableRuns]; ok { + t.Fatalf("marker has an entry for runs before it was ever marked: %+v", marker) + } +} + +// TestPreviewMarkerStore_PerTableNotGlobal proves the marker is per swept +// table, not per database: marking one table previewed must not mark a +// sibling table previewed too (AC-OFFICE-RUN-HISTORY-RETENTION-003.4). +func TestPreviewMarkerStore_PerTableNotGlobal(t *testing.T) { + _, raw := newTestSettingsStore(t) + store := NewPreviewMarkerStore(raw) + ctx := context.Background() + + if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, time.Now().UTC()); err != nil { + t.Fatalf("MarkCompleted(office_routine_runs): %v", err) + } + + marker, _ := store.Get(ctx) + if _, ok := marker[TableRuns]; ok { + t.Fatalf("marking office_routine_runs previewed also marked runs: %+v", marker) + } + + if err := store.MarkCompleted(ctx, TableRuns, time.Now().UTC()); err != nil { + t.Fatalf("MarkCompleted(runs): %v", err) + } + marker, _ = store.Get(ctx) + if len(marker) != 2 { + t.Fatalf("marker after both tables previewed = %+v, want 2 entries", marker) + } +} + +// TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed proves +// AC-OFFICE-RUN-HISTORY-RETENTION-003.10: a present-but-corrupt marker +// document is treated as "not yet previewed" for every swept table, which +// is the safe direction because a spurious re-preview deletes nothing. +func TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed(t *testing.T) { + _, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable marker: %v", err) + } + + store := NewPreviewMarkerStore(raw) + marker, readable := store.Get(ctx) + if readable { + t.Fatalf("Get() readable = true for an unparseable marker, want false") + } + if len(marker) != 0 { + t.Fatalf("Get() marker = %+v on unparseable document, want empty (not-previewed)", marker) + } +} + +// TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable proves that +// marking a table previewed after a corrupt document is detected replaces +// the corrupt document rather than erroring forever. +func TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable(t *testing.T) { + _, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable marker: %v", err) + } + store := NewPreviewMarkerStore(raw) + + at := time.Now().UTC() + if err := store.MarkCompleted(ctx, TableRuns, at); err != nil { + t.Fatalf("MarkCompleted after corrupt document: %v", err) + } + marker, readable := store.Get(ctx) + if !readable { + t.Fatalf("Get() readable = false after a fresh valid write") + } + if _, ok := marker[TableRuns]; !ok { + t.Fatalf("marker missing runs entry after recovery write: %+v", marker) + } +} diff --git a/apps/backend/internal/office/retention/runtime.go b/apps/backend/internal/office/retention/runtime.go new file mode 100644 index 00000000000..06af274b44a --- /dev/null +++ b/apps/backend/internal/office/retention/runtime.go @@ -0,0 +1,61 @@ +package retention + +import ( + "context" + + "github.com/kandev/kandev/internal/db" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// Runtime composes every retention component behind one Start/Stop pair, +// mirroring internal/system/storage.Runtime's shape: a scheduler goroutine, +// an HTTP handler, and a health checker, wired together once at boot over +// the shared pool and the shared key/value settings store. +type Runtime struct { + Store *Store + SettingsStore *SettingsStore + PreviewMarker *PreviewMarkerStore + Sweeper *Sweeper + Scheduler *Scheduler + Checker *Checker + Handler *Handler +} + +// NewRuntime constructs every retention component. logError, when set, is +// forwarded to Handler for a failed settings read or write; it never +// affects the sweep, which fails closed on its own terms (see +// SettingsStore.GetSettingsForSweep). +func NewRuntime(pool *db.Pool, settingsStore *systemsettings.Store, logError func(string, error)) *Runtime { + store := NewStore(pool) + retentionSettings := NewSettingsStore(settingsStore) + previewMarker := NewPreviewMarkerStore(settingsStore) + sweeper := NewSweeper(pool, store, retentionSettings, previewMarker) + scheduler := NewScheduler(retentionSettings, sweeper, SchedulerOptions{}) + checker := NewChecker(retentionSettings, sweeper, previewMarker) + handler := NewHandler(HandlerConfig{ + SettingsStore: retentionSettings, + Sweeper: sweeper, + OnSettingsChanged: scheduler.ApplySettings, + LogError: logError, + }) + return &Runtime{ + Store: store, + SettingsStore: retentionSettings, + PreviewMarker: previewMarker, + Sweeper: sweeper, + Scheduler: scheduler, + Checker: checker, + Handler: handler, + } +} + +// Start begins the scheduler loop (census immediately, sweep gated by +// enablement — see scheduler.go). Idempotent: safe to call once at boot. +func (r *Runtime) Start(ctx context.Context) error { + return r.Scheduler.Start(ctx) +} + +// Stop cancels and joins the scheduler loop. +func (r *Runtime) Stop() { + r.Scheduler.Stop() +} diff --git a/apps/backend/internal/office/retention/runtime_test.go b/apps/backend/internal/office/retention/runtime_test.go new file mode 100644 index 00000000000..684097e4a5c --- /dev/null +++ b/apps/backend/internal/office/retention/runtime_test.go @@ -0,0 +1,79 @@ +package retention + +import ( + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +func newTestRuntime(t *testing.T) *Runtime { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsStore, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + return NewRuntime(pool, settingsStore, nil) +} + +func TestNewRuntime_WiresEveryComponentNonNil(t *testing.T) { + runtime := newTestRuntime(t) + if runtime.Store == nil || runtime.SettingsStore == nil || runtime.PreviewMarker == nil || + runtime.Sweeper == nil || runtime.Scheduler == nil || runtime.Checker == nil || runtime.Handler == nil { + t.Fatalf("Runtime has a nil component: %+v", runtime) + } +} + +func TestRuntime_StartStopIsClean(t *testing.T) { + runtime := newTestRuntime(t) + ctx := t.Context() + + if err := runtime.Start(ctx); err != nil { + t.Fatalf("Start: %v", err) + } + defer runtime.Stop() + + if err := runtime.Start(ctx); err != nil { + t.Fatalf("second Start: %v", err) + } + runtime.Stop() + runtime.Stop() // idempotent +} + +func TestRuntime_HandlerOnSettingsChangedReachesScheduler(t *testing.T) { + runtime := newTestRuntime(t) + ctx := t.Context() + if err := runtime.Start(ctx); err != nil { + t.Fatalf("Start: %v", err) + } + defer runtime.Stop() + + settings := DefaultSettings() + settings.SweepIntervalHours = 2 + saved, err := runtime.SettingsStore.SaveSettings(ctx, settings) + if err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + // Handler's OnSettingsChanged is wired to Scheduler.ApplySettings; call + // it exactly as putRetention would and confirm the running scheduler + // picked up the change (AC-004.5, without a restart). + runtime.Handler.config.OnSettingsChanged(saved) + if got := runtime.Scheduler.latestSettings(); got.SweepIntervalHours != 2 { + t.Fatalf("Scheduler.latestSettings().SweepIntervalHours = %d, want 2", got.SweepIntervalHours) + } +} diff --git a/apps/backend/internal/office/retention/scheduler.go b/apps/backend/internal/office/retention/scheduler.go new file mode 100644 index 00000000000..ec0725783eb --- /dev/null +++ b/apps/backend/internal/office/retention/scheduler.go @@ -0,0 +1,192 @@ +package retention + +import ( + "context" + "sync" + "time" +) + +// firstSweepDelay is the fixed delay before the first sweep after Start, +// and after retention transitions from disabled to enabled +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.10, -002.13). Arming at the full +// interval instead — as the census timer does — would mean an install +// restarted more often than the interval never sweeps at all. +const firstSweepDelay = 5 * time.Minute + +// SchedulerOptions configures Scheduler construction. After is injectable +// for deterministic tests; production leaves it nil and gets time.After. +type SchedulerOptions struct { + After func(time.Duration) <-chan time.Time +} + +// Scheduler owns two independent timers on one goroutine, modelled on +// internal/system/storage.Scheduler: +// +// - The census timer always runs, on the configured sweep interval, +// whether or not retention is enabled (AC-OFFICE-RUN-HISTORY-RETENTION-003.11): +// a disabled install is exactly the one whose tables grow unattended, +// and the only one where an unrecognized status would otherwise never +// be noticed. +// - The sweep timer only runs while enabled, armed at firstSweepDelay +// after Start or after a disabled-to-enabled transition, and at the +// full interval thereafter (AC-OFFICE-RUN-HISTORY-RETENTION-002.10, +// -002.13). +// +// A settings change wakes the loop and re-arms both timers from the +// moment of the change (fixed-delay, not fixed-rate): the sweep timer at +// firstSweepDelay only when retention just turned on, otherwise at the +// (possibly new) full interval; the census timer always at the full +// interval. The census timer also refreshes the shared settings record, so a +// backend that did not serve a settings write still adopts it while disabled. +type Scheduler struct { + settingsStore *SettingsStore + sweeper *Sweeper + after func(time.Duration) <-chan time.Time + + lifecycleMu sync.Mutex + mu sync.Mutex + cancel context.CancelFunc + wake chan struct{} + latest Settings + wg sync.WaitGroup +} + +// NewScheduler wires the scheduler to its dependencies. +func NewScheduler(settingsStore *SettingsStore, sweeper *Sweeper, options SchedulerOptions) *Scheduler { + after := options.After + if after == nil { + after = time.After + } + return &Scheduler{settingsStore: settingsStore, sweeper: sweeper, after: after} +} + +// Start begins the scheduler loop. A no-op when already running. +func (s *Scheduler) Start(ctx context.Context) error { + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + s.mu.Lock() + running := s.cancel != nil + s.mu.Unlock() + if running { + return nil + } + + // GetSettings never fails outright: an unreadable or unparseable stored + // document still yields DefaultSettings, so the scheduler starts on + // those defaults rather than never starting at all. The wrapped error + // is reported separately through Checker.Check. + settings, _ := s.settingsStore.GetSettings(ctx) + + s.mu.Lock() + workerCtx, cancel := context.WithCancel(ctx) + s.cancel = cancel + s.wake = make(chan struct{}, 1) + s.latest = settings + s.wg.Add(1) + wake := s.wake + s.mu.Unlock() + + go s.run(workerCtx, settings, wake) + return nil +} + +// ApplySettings notifies a running scheduler that settings changed, +// re-arming both timers from this moment. A no-op when not running. +func (s *Scheduler) ApplySettings(settings Settings) { + s.mu.Lock() + if s.cancel == nil || s.wake == nil { + s.mu.Unlock() + return + } + s.latest = settings + wake := s.wake + s.mu.Unlock() + select { + case wake <- struct{}{}: + default: + } +} + +// Stop cancels the scheduler loop and joins it. A no-op when not running. +func (s *Scheduler) Stop() { + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + s.mu.Lock() + cancel := s.cancel + s.cancel = nil + s.wake = nil + s.mu.Unlock() + if cancel != nil { + cancel() + s.wg.Wait() + } +} + +func (s *Scheduler) run(ctx context.Context, settings Settings, wake <-chan struct{}) { + defer s.wg.Done() + + // The first census evaluation runs here, on the scheduler goroutine, + // not blocking Start's caller (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's + // "runs at Start, not at the first sweep"). + s.sweeper.RunCensus(ctx) + census := s.after(sweepInterval(settings)) + + var sweep <-chan time.Time + if settings.Enabled { + sweep = s.after(firstSweepDelay) + } + + for { + select { + case <-ctx.Done(): + return + case <-wake: + wasEnabled := settings.Enabled + settings = s.latestSettings() + + sweep = nil + if settings.Enabled { + if wasEnabled { + sweep = s.after(sweepInterval(settings)) + } else { + sweep = s.after(firstSweepDelay) + } + } + census = s.after(sweepInterval(settings)) + + case <-census: + s.sweeper.RunCensus(ctx) + if latest, err := s.settingsStore.GetSettings(ctx); err == nil && latest != settings { + wasEnabled := settings.Enabled + settings = latest + s.mu.Lock() + s.latest = latest + s.mu.Unlock() + + sweep = nil + if settings.Enabled { + if wasEnabled { + sweep = s.after(sweepInterval(settings)) + } else { + sweep = s.after(firstSweepDelay) + } + } + } + census = s.after(sweepInterval(settings)) + + case <-sweep: + s.sweeper.RunSweep(ctx) + sweep = s.after(sweepInterval(settings)) + } + } +} + +func (s *Scheduler) latestSettings() Settings { + s.mu.Lock() + defer s.mu.Unlock() + return s.latest +} + +func sweepInterval(settings Settings) time.Duration { + return time.Duration(settings.SweepIntervalHours) * time.Hour +} diff --git a/apps/backend/internal/office/retention/scheduler_test.go b/apps/backend/internal/office/retention/scheduler_test.go new file mode 100644 index 00000000000..b790497c5a9 --- /dev/null +++ b/apps/backend/internal/office/retention/scheduler_test.go @@ -0,0 +1,351 @@ +package retention + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// fakeAfter is a deterministic stand-in for time.After, keyed by duration: +// every call for the same duration returns the same channel, so a test can +// fire a specific timer (e.g. firstSweepDelay vs. the configured interval) +// without racing real wall-clock time. armed records every duration the +// scheduler has requested, in order, so a test can prove which timer was +// armed and when — including proving RunCensus already completed +// synchronously before the following after() call. +type fakeAfter struct { + mu sync.Mutex + chans map[time.Duration]chan time.Time + armed chan time.Duration +} + +func newFakeAfter() *fakeAfter { + return &fakeAfter{chans: map[time.Duration]chan time.Time{}, armed: make(chan time.Duration, 64)} +} + +func (f *fakeAfter) after(d time.Duration) <-chan time.Time { + f.mu.Lock() + ch, ok := f.chans[d] + if !ok { + ch = make(chan time.Time, 1) + f.chans[d] = ch + } + f.mu.Unlock() + f.armed <- d + return ch +} + +func (f *fakeAfter) fire(t *testing.T, d time.Duration) { + t.Helper() + f.mu.Lock() + ch, ok := f.chans[d] + f.mu.Unlock() + if !ok { + t.Fatalf("fire: no timer ever armed for duration %v", d) + } + ch <- time.Now() +} + +// waitArmed blocks until after() has been called with duration d, +// draining (and discarding) any other durations seen along the way. +func (f *fakeAfter) waitArmed(t *testing.T, d time.Duration) { + t.Helper() + deadline := time.After(2 * time.Second) + for { + select { + case got := <-f.armed: + if got == d { + return + } + case <-deadline: + t.Fatalf("timed out waiting for a timer to be armed at %v", d) + } + } +} + +// assertNotArmed drains any pending arm notifications and fails if d is +// among them. +func (f *fakeAfter) assertNotArmed(t *testing.T, d time.Duration) { + t.Helper() + for { + select { + case got := <-f.armed: + if got == d { + t.Fatalf("timer armed at %v, want it never armed", d) + } + default: + return + } + } +} + +func newTestScheduler(t *testing.T, settings Settings, fake *fakeAfter) (*Scheduler, *Sweeper) { + t.Helper() + sweeper, _ := newTestSweeper(t) + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + scheduler := NewScheduler(sweeper.settingsStore, sweeper, SchedulerOptions{After: fake.after}) + return scheduler, sweeper +} + +func TestScheduler_RunsCensusAtStartRegardlessOfEnabled(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + // The next after() call is census's re-arm, issued only once RunCensus + // (called synchronously beforehand in run()) has returned. + fake.waitArmed(t, sweepInterval(settings)) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("office_routine_runs census state = %v, want fresh (disabled must not block the census)", counts.OfficeRoutineRuns.State) + } + if counts.Runs.State != CensusFresh || counts.RunEvents.State != CensusFresh { + t.Fatalf("census not fresh for every table: %+v", counts) + } +} + +func TestScheduler_SweepArmedAtFirstDelayWhenEnabledAtStart(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, firstSweepDelay) +} + +func TestScheduler_SweepNotArmedWhenDisabledAtStart(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, sweepInterval(settings)) // census's arm proves the loop is running + fake.assertNotArmed(t, firstSweepDelay) +} + +func TestScheduler_SweepFiresAndReArmsAtFullIntervalNotFirstDelayAgain(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + settings.SweepIntervalHours = 1 + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + fake.waitArmed(t, firstSweepDelay) + fake.fire(t, firstSweepDelay) + + fake.waitArmed(t, sweepInterval(settings)) // re-armed at the full interval, not another 5-minute delay + + if _, ok := sweeper.LastSweepSnapshot(); !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true (the fired timer must have run a sweep)") + } +} + +func TestScheduler_EnablingFromDisabledArmsAtFirstDelay(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + enabled := settings + enabled.Enabled = true + scheduler.ApplySettings(enabled) + + fake.waitArmed(t, firstSweepDelay) +} + +func TestScheduler_SettingsChangeWhileEnabledReArmsAtNewIntervalNotFirstDelay(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + settings.SweepIntervalHours = 1 + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, firstSweepDelay) // initial arm; timer never fired + + changed := settings + changed.SweepIntervalHours = 2 + scheduler.ApplySettings(changed) + + fake.waitArmed(t, sweepInterval(changed)) // re-armed at the new interval, not firstSweepDelay again +} + +func TestScheduler_DisablingStopsArmingSweepButNotCensus(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, firstSweepDelay) + + disabled := settings + disabled.Enabled = false + scheduler.ApplySettings(disabled) + + fake.waitArmed(t, sweepInterval(disabled)) // census's re-arm proves the wake was processed + fake.assertNotArmed(t, firstSweepDelay) +} + +func TestScheduler_ReconcilesSharedSettingsOnCensus(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + settings.SweepIntervalHours = 1 + scheduler, sweeper := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + updated := settings + updated.Enabled = true + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), updated); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + fake.fire(t, sweepInterval(settings)) + + // The census timer is also the periodic shared-settings reconciliation + // point. Enabling retention in another backend must arm this process's + // first sweep even when no local PUT delivered ApplySettings. + fake.waitArmed(t, firstSweepDelay) +} + +func TestScheduler_StartTwiceIsNoop(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = false + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("first Start: %v", err) + } + defer scheduler.Stop() + fake.waitArmed(t, sweepInterval(settings)) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("second Start: %v", err) + } + // A second run() goroutine would double-arm; draining once more must + // time out rather than find another immediate arm. + select { + case d := <-fake.armed: + t.Fatalf("second Start armed another timer at %v", d) + case <-time.After(100 * time.Millisecond): + } +} + +// TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable proves +// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: an unreadable stored settings +// document must not disable retention silently. GetSettings already falls +// back to DefaultSettings on such a document (see settings_store_test.go); +// this proves Start actually uses that fallback and runs the loop instead +// of aborting before the goroutine ever spawns. +func TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable(t *testing.T) { + fake := newFakeAfter() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + if err := settingsRaw.Save(context.Background(), settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + settingsStore := NewSettingsStore(settingsRaw) + sweeper := NewSweeper(pool, NewStore(pool), settingsStore, NewPreviewMarkerStore(settingsRaw)) + scheduler := NewScheduler(settingsStore, sweeper, SchedulerOptions{After: fake.after}) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + defer scheduler.Stop() + + // DefaultSettings().Enabled is true, so both timers must arm off the + // fallback defaults rather than the loop never starting at all. + fake.waitArmed(t, sweepInterval(DefaultSettings())) + fake.waitArmed(t, firstSweepDelay) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh { + t.Fatalf("office_routine_runs census state = %v, want fresh (Start must run the census off defaults)", counts.OfficeRoutineRuns.State) + } +} + +func TestScheduler_StopJoinsWithoutHanging(t *testing.T) { + fake := newFakeAfter() + settings := DefaultSettings() + settings.Enabled = true + scheduler, _ := newTestScheduler(t, settings, fake) + + if err := scheduler.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + fake.waitArmed(t, firstSweepDelay) + + done := make(chan struct{}) + go func() { + scheduler.Stop() + close(done) + }() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("Stop did not return") + } +} diff --git a/apps/backend/internal/office/retention/settings_store.go b/apps/backend/internal/office/retention/settings_store.go new file mode 100644 index 00000000000..d7cf6d53cb5 --- /dev/null +++ b/apps/backend/internal/office/retention/settings_store.go @@ -0,0 +1,91 @@ +package retention + +import ( + "context" + "encoding/json" + "fmt" + + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// settingsKey is the internal/system/settings.Store key holding the +// retention policy document. The live key/value table is "settings", +// reached through settings.Store; system_settings is a legacy SQLite-only +// table read once for migration and is absent on PostgreSQL. +const settingsKey = "office_run_retention" + +// SettingsStore persists and reads the retention policy document. +type SettingsStore struct { + store *systemsettings.Store +} + +// NewSettingsStore wraps the shared key/value settings store. +func NewSettingsStore(store *systemsettings.Store) *SettingsStore { + return &SettingsStore{store: store} +} + +// GetSettings reads the retention policy for reporting and startup use. An +// unreadable or unparseable document falls back to DefaultSettings and +// wraps ErrInvalidPersistedSettings so the caller can raise a health issue; +// it never fails outright (AC-OFFICE-RUN-HISTORY-RETENTION-004.4). +func (s *SettingsStore) GetSettings(ctx context.Context) (Settings, error) { + raw, found, err := s.store.Get(ctx, settingsKey) + if err != nil { + return DefaultSettings(), fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + if !found { + return DefaultSettings(), nil + } + return decodeAndNormalize(raw, DefaultSettings()) +} + +// GetSettingsForSweep reads the retention policy on the writer pool at +// sweep start, per AC-OFFICE-RUN-HISTORY-RETENTION-004.5: a sweep must read +// the stored settings at its start rather than a value cached from a +// notification, so a backend that did not serve the write still sweeps +// under the new policy. Unlike GetSettings, a read or parse failure here +// returns ErrInvalidPersistedSettings with no usable Settings value — the +// caller must skip the sweep rather than fall back to the (possibly +// shorter) default window and delete history the operator configured the +// system to keep. A document that was never saved is the legitimate empty +// state, not a failure, and yields the defaults. +func (s *SettingsStore) GetSettingsForSweep(ctx context.Context) (Settings, error) { + raw, found, err := s.store.GetConsistent(ctx, settingsKey) + if err != nil { + return Settings{}, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + if !found { + return DefaultSettings(), nil + } + return decodeAndNormalize(raw, Settings{}) +} + +func decodeAndNormalize(raw []byte, fallback Settings) (Settings, error) { + var doc Settings + if err := json.Unmarshal(raw, &doc); err != nil { + return fallback, fmt.Errorf("%w: decode JSON: %w", ErrInvalidPersistedSettings, err) + } + normalized, err := NormalizeSettings(doc) + if err != nil { + return fallback, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err) + } + return normalized, nil +} + +// SaveSettings normalizes and persists a full settings document, replacing +// whatever was stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.9). A rejected +// write leaves the stored document unchanged. +func (s *SettingsStore) SaveSettings(ctx context.Context, in Settings) (Settings, error) { + normalized, err := NormalizeSettings(in) + if err != nil { + return Settings{}, err + } + raw, err := json.Marshal(normalized) + if err != nil { + return Settings{}, fmt.Errorf("encode retention settings: %w", err) + } + if err := s.store.Save(ctx, settingsKey, raw); err != nil { + return Settings{}, err + } + return normalized, nil +} diff --git a/apps/backend/internal/office/retention/settings_store_test.go b/apps/backend/internal/office/retention/settings_store_test.go new file mode 100644 index 00000000000..eacfb89e789 --- /dev/null +++ b/apps/backend/internal/office/retention/settings_store_test.go @@ -0,0 +1,154 @@ +package retention + +import ( + "context" + "errors" + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +func newTestSettingsStore(t *testing.T) (*SettingsStore, *systemsettings.Store) { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + raw, err := systemsettings.NewStore(db.NewPool(conn, conn)) + if err != nil { + t.Fatalf("new settings store: %v", err) + } + return NewSettingsStore(raw), raw +} + +func TestSettingsStore_MissingReturnsDefaults(t *testing.T) { + store, _ := newTestSettingsStore(t) + got, err := store.GetSettings(context.Background()) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettings() = %+v, want defaults", got) + } +} + +func TestSettingsStore_SaveThenGetRoundTrips(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.SweepIntervalHours = 12 + in.RoutineRuns.WindowDays = 90 + + saved, err := store.SaveSettings(ctx, in) + if err != nil { + t.Fatalf("SaveSettings: %v", err) + } + if saved != in { + t.Fatalf("SaveSettings returned %+v, want %+v", saved, in) + } + + got, err := store.GetSettings(ctx) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got != in { + t.Fatalf("GetSettings() = %+v, want %+v", got, in) + } +} + +func TestSettingsStore_SaveRejectsOutOfRangeAndLeavesStoredUnchanged(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.SweepIntervalHours = 12 + if _, err := store.SaveSettings(ctx, in); err != nil { + t.Fatalf("seed SaveSettings: %v", err) + } + + bad := DefaultSettings() + bad.BatchLimit = 1 + if _, err := store.SaveSettings(ctx, bad); !errors.Is(err, ErrValidation) { + t.Fatalf("SaveSettings(bad): err = %v, want ErrValidation", err) + } + + got, err := store.GetSettings(ctx) + if err != nil { + t.Fatalf("GetSettings: %v", err) + } + if got.SweepIntervalHours != 12 { + t.Fatalf("stored settings changed after rejected write: %+v", got) + } +} + +// TestSettingsStore_UnparseableFallsBackToDefaultsForReporting proves +// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: reading for reporting/startup +// tolerates an unreadable document and yields the documented defaults +// (wrapped in ErrInvalidPersistedSettings so a caller can raise a health +// issue), never failing outright. +func TestSettingsStore_UnparseableFallsBackToDefaultsForReporting(t *testing.T) { + store, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + got, err := store.GetSettings(ctx) + if !errors.Is(err, ErrInvalidPersistedSettings) { + t.Fatalf("GetSettings: err = %v, want ErrInvalidPersistedSettings", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettings() = %+v, want defaults on unparseable document", got) + } +} + +// TestSettingsStore_ForSweepFailsClosedOnUnparseable proves the other half +// of AC-OFFICE-RUN-HISTORY-RETENTION-004.5: reading settings *to delete by* +// must not silently fall back to the (possibly shorter) default window. The +// caller sees ErrInvalidPersistedSettings and a zero Settings value, and is +// expected to skip the sweep rather than run it under defaults. +func TestSettingsStore_ForSweepFailsClosedOnUnparseable(t *testing.T) { + store, raw := newTestSettingsStore(t) + ctx := context.Background() + if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + _, err := store.GetSettingsForSweep(ctx) + if !errors.Is(err, ErrInvalidPersistedSettings) { + t.Fatalf("GetSettingsForSweep: err = %v, want ErrInvalidPersistedSettings", err) + } +} + +func TestSettingsStore_ForSweepMissingUsesDefaults(t *testing.T) { + store, _ := newTestSettingsStore(t) + got, err := store.GetSettingsForSweep(context.Background()) + if err != nil { + t.Fatalf("GetSettingsForSweep: %v", err) + } + if got != DefaultSettings() { + t.Fatalf("GetSettingsForSweep() = %+v, want defaults when nothing stored", got) + } +} + +func TestSettingsStore_ForSweepReadsWriterPool(t *testing.T) { + store, _ := newTestSettingsStore(t) + ctx := context.Background() + in := DefaultSettings() + in.RoutineRuns.WindowDays = 3650 + if _, err := store.SaveSettings(ctx, in); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + got, err := store.GetSettingsForSweep(ctx) + if err != nil { + t.Fatalf("GetSettingsForSweep: %v", err) + } + if got.RoutineRuns.WindowDays != 3650 { + t.Fatalf("GetSettingsForSweep() = %+v, want the just-saved window", got) + } +} diff --git a/apps/backend/internal/office/retention/settings_wire.go b/apps/backend/internal/office/retention/settings_wire.go new file mode 100644 index 00000000000..73af2563374 --- /dev/null +++ b/apps/backend/internal/office/retention/settings_wire.go @@ -0,0 +1,181 @@ +package retention + +import ( + "bytes" + "encoding/json" + "fmt" + "io" +) + +// retentionSettingsEnvelope captures each top-level field as raw JSON +// rather than a typed value, so a request body can be told apart into +// three cases per field: absent (nil RawMessage, take the documented +// default), present as the literal JSON null (rejected — +// AC-OFFICE-RUN-HISTORY-RETENTION-004.9), or present with a value (decoded +// and validated). A plain typed struct cannot distinguish the first two: an +// ordinary `*int` field is nil either way. +type retentionSettingsEnvelope struct { + Enabled json.RawMessage `json:"enabled"` + SweepIntervalHours json.RawMessage `json:"sweep_interval_hours"` + BatchLimit json.RawMessage `json:"batch_limit"` + RoutineRuns json.RawMessage `json:"routine_runs"` + Runs json.RawMessage `json:"runs"` + RunEvents json.RawMessage `json:"run_events"` +} + +type tableSettingsEnvelope struct { + WindowDays json.RawMessage `json:"window_days"` + FloorPerOwner json.RawMessage `json:"floor_per_owner"` + WarnRows json.RawMessage `json:"warn_rows"` +} + +type runEventsSettingsEnvelope struct { + WarnRows json.RawMessage `json:"warn_rows"` +} + +// decodeRetentionSettings implements AC-OFFICE-RUN-HISTORY-RETENTION-004.9's +// PUT semantics: a full replace where an omitted field takes its documented +// default, an unrecognized field or an explicit null is rejected naming the +// field, and nothing is written on any rejection (the caller is expected to +// not persist the zero-value Settings returned alongside a non-nil error). +// Range validation (AC-004.3) is NormalizeSettings's job, called by +// SettingsStore.SaveSettings after this decode succeeds. +func decodeRetentionSettings(body []byte) (Settings, error) { + defaults := DefaultSettings() + trimmed := bytes.TrimSpace(body) + if len(trimmed) == 0 || trimmed[0] != '{' { + return Settings{}, fmt.Errorf("request body: expected a JSON object") + } + + dec := json.NewDecoder(bytes.NewReader(trimmed)) + dec.DisallowUnknownFields() + var env retentionSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return Settings{}, err + } + var extra any + if err := dec.Decode(&extra); err != io.EOF { + if err == nil { + return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object") + } + return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object: %w", err) + } + + enabled, err := decodeBoolField(env.Enabled, "enabled", defaults.Enabled) + if err != nil { + return Settings{}, err + } + sweepIntervalHours, err := decodeIntField(env.SweepIntervalHours, "sweep_interval_hours", defaults.SweepIntervalHours) + if err != nil { + return Settings{}, err + } + batchLimit, err := decodeIntField(env.BatchLimit, "batch_limit", defaults.BatchLimit) + if err != nil { + return Settings{}, err + } + routineRuns, err := decodeTableSettings(env.RoutineRuns, "routine_runs", defaults.RoutineRuns) + if err != nil { + return Settings{}, err + } + runs, err := decodeTableSettings(env.Runs, "runs", defaults.Runs) + if err != nil { + return Settings{}, err + } + runEvents, err := decodeRunEventsSettings(env.RunEvents, "run_events", defaults.RunEvents) + if err != nil { + return Settings{}, err + } + + return Settings{ + Enabled: enabled, + SweepIntervalHours: sweepIntervalHours, + BatchLimit: batchLimit, + RoutineRuns: routineRuns, + Runs: runs, + RunEvents: runEvents, + }, nil +} + +func decodeTableSettings(raw json.RawMessage, name string, defaults TableSettings) (TableSettings, error) { + if raw == nil { + return defaults, nil + } + if isJSONNull(raw) { + return TableSettings{}, rejectNull(name) + } + dec := json.NewDecoder(bytes.NewReader(raw)) + dec.DisallowUnknownFields() + var env tableSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return TableSettings{}, fmt.Errorf("%s: %w", name, err) + } + windowDays, err := decodeIntField(env.WindowDays, name+".window_days", defaults.WindowDays) + if err != nil { + return TableSettings{}, err + } + floorPerOwner, err := decodeIntField(env.FloorPerOwner, name+".floor_per_owner", defaults.FloorPerOwner) + if err != nil { + return TableSettings{}, err + } + warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows) + if err != nil { + return TableSettings{}, err + } + return TableSettings{WindowDays: windowDays, FloorPerOwner: floorPerOwner, WarnRows: warnRows}, nil +} + +func decodeRunEventsSettings(raw json.RawMessage, name string, defaults RunEventsSettings) (RunEventsSettings, error) { + if raw == nil { + return defaults, nil + } + if isJSONNull(raw) { + return RunEventsSettings{}, rejectNull(name) + } + dec := json.NewDecoder(bytes.NewReader(raw)) + dec.DisallowUnknownFields() + var env runEventsSettingsEnvelope + if err := dec.Decode(&env); err != nil { + return RunEventsSettings{}, fmt.Errorf("%s: %w", name, err) + } + warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows) + if err != nil { + return RunEventsSettings{}, err + } + return RunEventsSettings{WarnRows: warnRows}, nil +} + +func decodeBoolField(raw json.RawMessage, name string, defaultVal bool) (bool, error) { + if raw == nil { + return defaultVal, nil + } + if isJSONNull(raw) { + return false, rejectNull(name) + } + var v bool + if err := json.Unmarshal(raw, &v); err != nil { + return false, fmt.Errorf("%s: %w", name, err) + } + return v, nil +} + +func decodeIntField(raw json.RawMessage, name string, defaultVal int) (int, error) { + if raw == nil { + return defaultVal, nil + } + if isJSONNull(raw) { + return 0, rejectNull(name) + } + var v int + if err := json.Unmarshal(raw, &v); err != nil { + return 0, fmt.Errorf("%s: %w", name, err) + } + return v, nil +} + +func isJSONNull(raw json.RawMessage) bool { + return bytes.Equal(bytes.TrimSpace(raw), []byte("null")) +} + +func rejectNull(name string) error { + return fmt.Errorf("%s: must not be null; omit the field to use its default", name) +} diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go new file mode 100644 index 00000000000..52d80b8b138 --- /dev/null +++ b/apps/backend/internal/office/retention/store.go @@ -0,0 +1,476 @@ +package retention + +import ( + "context" + "database/sql" + "errors" + "fmt" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/db/dialect" +) + +// queryer is the subset of *sqlx.DB / *sqlx.Conn this package needs to run +// a sweep or a census. The sweep runs every statement for its whole +// duration through one queryer: the writer pool directly on SQLite, or one +// dedicated connection on PostgreSQL (see lock.go) — so the advisory lock +// and the batches it protects can never diverge onto different sessions. +type queryer interface { + db.Rebinder + ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error) + QueryxContext(ctx context.Context, query string, args ...any) (*sqlx.Rows, error) + QueryRowxContext(ctx context.Context, query string, args ...any) *sqlx.Row + GetContext(ctx context.Context, dest any, query string, args ...any) error + SelectContext(ctx context.Context, dest any, query string, args ...any) error + BeginTxx(ctx context.Context, opts *sql.TxOptions) (*sqlx.Tx, error) +} + +// Store is the raw SQL access layer for retention: the eligibility counts +// (used by both the preview and the backlog check), the batch deletes, and +// the status census. It holds no state of its own. +type Store struct { + pool *db.Pool +} + +// NewStore wraps the database pool office_routine_runs, runs, and their +// satellites live in. +func NewStore(pool *db.Pool) *Store { + return &Store{pool: pool} +} + +// IsPostgres reports whether the writer pool is PostgreSQL. On SQLite one +// backend process owns the database file, so the in-process sweeping guard +// is sufficient and no advisory lock is taken. +func (s *Store) IsPostgres() bool { + return dialect.IsPostgres(s.pool.Writer().DriverName()) +} + +func routineRunEligibleSubquery() string { + return ` + SELECT id, + COALESCE(completed_at, created_at) AS completion_time, + ROW_NUMBER() OVER ( + PARTITION BY routine_id + ORDER BY COALESCE(completed_at, created_at) DESC, id DESC + ) AS rn + FROM office_routine_runs + WHERE status IN (?)` +} + +func runEligibleSubquery() string { + return ` + SELECT id, + COALESCE(finished_at, requested_at) AS completion_time, + ROW_NUMBER() OVER ( + PARTITION BY agent_profile_id + ORDER BY COALESCE(finished_at, requested_at) DESC, id DESC + ) AS rn + FROM runs + WHERE status IN (?) + AND NOT EXISTS ( + SELECT 1 + FROM office_agent_pause_recoveries + WHERE office_agent_pause_recoveries.failed_run_id = runs.id + )` +} + +// CountEligibleRoutineRuns is the office_routine_runs eligibility count, +// uncapped by any batch limit: used for both the preview's WouldDelete +// (AC-OFFICE-RUN-HISTORY-RETENTION-003.9) and the backlog determination +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3) so the two can never disagree +// about what "eligible" means. +func (s *Store) CountEligibleRoutineRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) { + query := `SELECT COUNT(*) FROM (` + routineRunEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?` + return countEligible(ctx, q, query, RoutineRunHistoryStatuses, floor, cutoff) +} + +// CountEligibleRuns is the runs table's equivalent of CountEligibleRoutineRuns. +func (s *Store) CountEligibleRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) { + query := `SELECT COUNT(*) FROM (` + runEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?` + return countEligible(ctx, q, query, RunHistoryStatuses, floor, cutoff) +} + +func countEligible(ctx context.Context, q queryer, query string, statuses []string, floor int, cutoff time.Time) (int64, error) { + bound, args, err := db.Bind(q, query, statuses, floor, cutoff) + if err != nil { + return 0, err + } + var count int64 + if err := q.QueryRowxContext(ctx, bound, args...).Scan(&count); err != nil { + return 0, err + } + return count, nil +} + +// DeleteRoutineRunsBatch deletes at most batchLimit eligible +// office_routine_runs rows, oldest first, re-asserting status, age and the +// per-owner floor in the same statement +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3, -002.4). One statement is +// already atomic, satisfying -002.5 without an explicit transaction. +func (s *Store) DeleteRoutineRunsBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (int64, error) { + query := ` + DELETE FROM office_routine_runs + WHERE id IN ( + SELECT id FROM (` + routineRunEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ORDER BY completion_time ASC, id ASC + LIMIT ? + ) + AND status IN (?) + AND COALESCE(completed_at, created_at) < ?` + bound, args, err := db.Bind(q, query, + RoutineRunHistoryStatuses, floor, cutoff, batchLimit, + RoutineRunHistoryStatuses, cutoff, + ) + if err != nil { + return 0, err + } + res, err := q.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + return res.RowsAffected() +} + +// RunBatchResult reports what one runs batch (and its satellites) actually +// deleted. Abandoned is true only when the batch was rolled back twice in a +// row and gave up — every count is then zero, matching what the rollback +// left committed (AC-OFFICE-RUN-HISTORY-RETENTION-002.7). +type RunBatchResult struct { + RunsDeleted int64 + RunEventsDeleted int64 + RouteAttemptsDeleted int64 + RunSkillsDeleted int64 + Abandoned bool +} + +// DeleteRunBatch selects up to batchLimit eligible runs, deletes each +// selected run's satellites and then the run itself in one transaction, +// re-asserting the whole eligibility predicate (status, age, and floor) at +// delete time. If a concurrent ScheduleRetry resurrects a row between +// selection and delete, step 5's affected-row count falls short of the +// selected id count; the whole transaction is rolled back and retried once +// with a fresh selection. A second mismatch abandons the batch +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4, -002.7). +func (s *Store) DeleteRunBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (RunBatchResult, error) { + for attempt := 0; attempt < 2; attempt++ { + if testBeforeSelectEligibleRunIDs != nil { + testBeforeSelectEligibleRunIDs(attempt) + } + ids, err := s.selectEligibleRunIDs(ctx, q, cutoff, floor, batchLimit) + if err != nil { + return RunBatchResult{}, err + } + if len(ids) == 0 { + return RunBatchResult{}, nil + } + if testAfterSelectEligibleRunIDs != nil { + testAfterSelectEligibleRunIDs(attempt, ids) + } + result, matched, err := s.deleteRunBatchOnce(ctx, q, ids, cutoff, floor) + if err != nil { + return RunBatchResult{}, err + } + if matched { + return result, nil + } + } + return RunBatchResult{Abandoned: true}, nil +} + +// testBeforeSelectEligibleRunIDs and testAfterSelectEligibleRunIDs, when +// set, bracket each of DeleteRunBatch's (at most two) selections — a +// deterministic seam this package's own tests use to exercise the +// selection-to-delete resurrection race, including the two-consecutive- +// mismatches abandon path, without depending on cross-connection goroutine +// timing. Never set outside tests. +var ( + testBeforeSelectEligibleRunIDs func(attempt int) + testAfterSelectEligibleRunIDs func(attempt int, ids []string) +) + +func (s *Store) selectEligibleRunIDs(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) ([]string, error) { + query := ` + SELECT id FROM (` + runEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ORDER BY completion_time ASC, id ASC + LIMIT ?` + bound, args, err := db.Bind(q, query, RunHistoryStatuses, floor, cutoff, batchLimit) + if err != nil { + return nil, err + } + rows, err := q.QueryxContext(ctx, bound, args...) + if err != nil { + return nil, err + } + defer func() { _ = rows.Close() }() + var ids []string + for rows.Next() { + var id string + if err := rows.Scan(&id); err != nil { + return nil, err + } + ids = append(ids, id) + } + return ids, rows.Err() +} + +func (s *Store) deleteRunBatchOnce(ctx context.Context, q queryer, ids []string, cutoff time.Time, floor int) (RunBatchResult, bool, error) { + tx, err := q.BeginTxx(ctx, nil) + if err != nil { + return RunBatchResult{}, false, err + } + committed := false + defer func() { + if !committed { + _ = tx.Rollback() + } + }() + + runEventsDeleted, err := deleteByRunIDs(ctx, tx, "run_events", ids) + if err != nil { + return RunBatchResult{}, false, err + } + routeAttemptsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_route_attempts", ids) + if err != nil { + return RunBatchResult{}, false, err + } + runSkillsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_skills", ids) + if err != nil { + return RunBatchResult{}, false, err + } + + runsDeleted, err := deleteRunsByIDs(ctx, tx, ids, cutoff, floor) + if err != nil { + return RunBatchResult{}, false, err + } + if runsDeleted != int64(len(ids)) { + return RunBatchResult{}, false, nil + } + + if err := tx.Commit(); err != nil { + return RunBatchResult{}, false, err + } + committed = true + return RunBatchResult{ + RunsDeleted: runsDeleted, + RunEventsDeleted: runEventsDeleted, + RouteAttemptsDeleted: routeAttemptsDeleted, + RunSkillsDeleted: runSkillsDeleted, + }, true, nil +} + +// retentionMaxHostParams caps id-list placeholders per statement. +// batch_limit's documented range (AC-OFFICE-RUN-HISTORY-RETENTION-004.3) +// permits up to 100,000, which would otherwise bind that many ids in one IN +// clause and can overflow SQLite's compiled variable-count limit or +// PostgreSQL's wire-protocol parameter cap. Matches the bound this repo +// already uses for the same reason (internal/task/repository/sqlite's +// sqliteMaxHostParams). +const retentionMaxHostParams = 500 + +// chunkIDs splits ids into sub-slices of at most size entries so an +// IN-clause query built from them stays under retentionMaxHostParams +// regardless of batch_limit. An empty input returns nil rather than one +// empty chunk, since an empty IN () clause is a SQL syntax error. +func chunkIDs(ids []string, size int) [][]string { + if len(ids) == 0 { + return nil + } + if size <= 0 || len(ids) <= size { + return [][]string{ids} + } + chunks := make([][]string, 0, (len(ids)+size-1)/size) + for i := 0; i < len(ids); i += size { + end := i + size + if end > len(ids) { + end = len(ids) + } + chunks = append(chunks, ids[i:end]) + } + return chunks +} + +// deleteRunsByIDs deletes ids from runs, in chunks of at most +// retentionMaxHostParams. Each chunk's outer WHERE re-asserts status and age +// directly, in addition to the floor-checking subquery: PostgreSQL's +// EvalPlanQual recheck of a concurrently updated row re-evaluates a direct +// column predicate against the row's fresh values, but does not rebuild an +// uncorrelated id-membership subquery, so the subquery alone is not enough +// to exclude a row resurrected between selection and delete +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4). +func deleteRunsByIDs(ctx context.Context, tx *sqlx.Tx, ids []string, cutoff time.Time, floor int) (int64, error) { + var total int64 + for _, chunk := range chunkIDs(ids, retentionMaxHostParams) { + query := ` + DELETE FROM runs + WHERE id IN (?) + AND id IN ( + SELECT id FROM (` + runEligibleSubquery() + `) ranked + WHERE rn > ? AND completion_time < ? + ) + AND status IN (?) + AND COALESCE(finished_at, requested_at) < ?` + bound, args, err := db.Bind(tx, query, chunk, RunHistoryStatuses, floor, cutoff, RunHistoryStatuses, cutoff) + if err != nil { + return 0, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + affected, err := res.RowsAffected() + if err != nil { + return 0, err + } + total += affected + } + return total, nil +} + +// deleteByRunIDs deletes run-id-keyed satellite rows for one table. table +// must be a hardcoded identifier from a call site in this package, never a +// caller-supplied string — it is interpolated directly into the query text. +func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) { + var total int64 + for _, chunk := range chunkIDs(ids, retentionMaxHostParams) { + query := fmt.Sprintf(`DELETE FROM %s WHERE run_id IN (?)`, table) + bound, args, err := db.Bind(tx, query, chunk) + if err != nil { + return 0, err + } + res, err := tx.ExecContext(ctx, bound, args...) + if err != nil { + return 0, err + } + affected, err := res.RowsAffected() + if err != nil { + return 0, err + } + total += affected + } + return total, nil +} + +// CountRunEvents is run_events' plain retained count: it has no status +// column, so it keeps a bare COUNT(*) rather than a census. +func (s *Store) CountRunEvents(ctx context.Context, q queryer) (int64, error) { + var count int64 + if err := q.GetContext(ctx, &count, `SELECT COUNT(*) FROM run_events`); err != nil { + return 0, err + } + return count, nil +} + +// CensusRoutineRuns issues office_routine_runs' status census +// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10, -003.5, -003.11). The +// unknown-status detector needs every distinct status value, which the +// top-routine attribution's aggregation does not carry, so it stays a +// separate GROUP BY status scan. The retained total and the top-routine +// attribution, by contrast, must agree with each other by construction — +// AC-003.5's share is defined as one routine's retained rows as a +// proportion of the table's retained count — so routineRunCensusTotals +// reads both from one statement rather than two, which a concurrent write +// between them could otherwise make disagree. +func (s *Store) CensusRoutineRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + statusCounts, err := statusCensus(ctx, q, "office_routine_runs") + if err != nil { + return TableCensus{}, err + } + _, unknown := summarizeStatusCensus(statusCounts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses) + + if testBetweenRoutineRunCensusReads != nil { + testBetweenRoutineRunCensusReads(q) + } + + retained, topID, topCount, err := routineRunCensusTotals(ctx, q) + if err != nil { + return TableCensus{}, err + } + census := TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now} + if retained > 0 { + census.TopRoutineID = topID + census.TopRoutineShare = float64(topCount) / float64(retained) + } + return census, nil +} + +// testBetweenRoutineRunCensusReads, when set, runs right after the +// unknown-status scan and right before routineRunCensusTotals' single- +// statement read — a deterministic seam for proving a concurrent write +// landing there cannot desynchronize the retained total from the +// top-routine attribution, since both now come from that one statement. +// Never set outside tests. +var testBetweenRoutineRunCensusReads func(q queryer) + +// CensusRuns issues runs' status census, the runs-table equivalent of +// CensusRoutineRuns without the routine attribution AC-003.5 is specific +// to office_routine_runs. +func (s *Store) CensusRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + statusCounts, err := statusCensus(ctx, q, "runs") + if err != nil { + return TableCensus{}, err + } + retained, unknown := summarizeStatusCensus(statusCounts, RunHistoryStatuses, RunLiveStatuses) + return TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now}, nil +} + +// CensusRunEvents is run_events' census: a plain count, since the table +// has no status column and therefore no unknown-status detection. +func (s *Store) CensusRunEvents(ctx context.Context, q queryer, now time.Time) (TableCensus, error) { + count, err := s.CountRunEvents(ctx, q) + if err != nil { + return TableCensus{}, err + } + return TableCensus{RetainedCount: count, AsOf: now}, nil +} + +func statusCensus(ctx context.Context, q queryer, table string) (map[string]int64, error) { + query := fmt.Sprintf(`SELECT status, COUNT(*) AS count FROM %s GROUP BY status`, table) + rows, err := q.QueryxContext(ctx, query) + if err != nil { + return nil, err + } + defer func() { _ = rows.Close() }() + counts := map[string]int64{} + for rows.Next() { + var status string + var count int64 + if err := rows.Scan(&status, &count); err != nil { + return nil, err + } + counts[status] = count + } + return counts, rows.Err() +} + +// routineRunCensusTotals reads office_routine_runs' table-wide retained +// total and the routine holding the largest share of it from one +// statement: a single GROUP BY routine_id pass, with the table total taken +// as a window sum over that same grouping, so the two numbers reflect +// exactly one snapshot and a share computed from them can never exceed 1.0. +// An empty table produces no groups at all; that is the legitimate zero +// state, not a failure, so sql.ErrNoRows is not propagated. +func routineRunCensusTotals(ctx context.Context, q queryer) (retained int64, topRoutineID string, topRoutineCount int64, err error) { + var row struct { + RoutineID string `db:"routine_id"` + Retained int64 `db:"retained"` + Total int64 `db:"total"` + } + query := ` + SELECT routine_id, COUNT(*) AS retained, SUM(COUNT(*)) OVER () AS total + FROM office_routine_runs + GROUP BY routine_id + ORDER BY retained DESC, routine_id ASC + LIMIT 1` + if err := q.GetContext(ctx, &row, query); err != nil { + if errors.Is(err, sql.ErrNoRows) { + return 0, "", 0, nil + } + return 0, "", 0, err + } + return row.Total, row.RoutineID, row.Retained, nil +} diff --git a/apps/backend/internal/office/retention/store_census_test.go b/apps/backend/internal/office/retention/store_census_test.go new file mode 100644 index 00000000000..a208d79aad8 --- /dev/null +++ b/apps/backend/internal/office/retention/store_census_test.go @@ -0,0 +1,235 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/kandev/kandev/internal/db" +) + +func TestCensusRoutineRuns_RetainedCountIsSumOfEveryStatus(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "skipped", timePtr(daysAgo(2)), daysAgo(2)) + seedRoutineRun(t, conn, newID(), "r-1", "received", nil, daysAgo(0)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3", census.RetainedCount) + } + if len(census.UnknownStatuses) != 0 { + t.Fatalf("unknownStatuses = %v, want none", census.UnknownStatuses) + } +} + +func TestCensusRoutineRuns_EmptyTableReturnsZeroNoTopRoutine(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 0 { + t.Fatalf("retainedCount = %d, want 0", census.RetainedCount) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID) + } +} + +func TestCensusRoutineRuns_DetectsUnknownStatus(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "quarantined" || census.UnknownStatuses[0].Count != 2 { + t.Fatalf("unknownStatuses = %v, want [{quarantined 2}]", census.UnknownStatuses) + } +} + +func TestCensusRoutineRuns_AttributesTopRoutineByRetainedShare(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-heavy") + seedRoutine(t, conn, "r-light") + for i := 0; i < 3; i++ { + seedRoutineRun(t, conn, newID(), "r-heavy", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + seedRoutineRun(t, conn, newID(), "r-light", "done", timePtr(daysAgo(1)), daysAgo(1)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 4 { + t.Fatalf("retainedCount = %d, want 4", census.RetainedCount) + } + if census.TopRoutineID != "r-heavy" { + t.Fatalf("topRoutineID = %q, want r-heavy", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 0.75; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } +} + +func TestCensusRoutineRuns_TiesAttributeToLowerRoutineID(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-b") + seedRoutine(t, conn, "r-a") + seedRoutineRun(t, conn, newID(), "r-b", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-a", "done", timePtr(daysAgo(1)), daysAgo(1)) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.TopRoutineID != "r-a" { + t.Fatalf("topRoutineID = %q, want r-a (lower id on tie)", census.TopRoutineID) + } +} + +func TestCensusRuns_RetainedCountIsSumOfEveryStatusNoTopAttribution(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRun(t, conn, newID(), "agent-1", "queued", nil, daysAgo(0)) + seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0)) + seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0)) + + census, err := store.CensusRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRuns: %v", err) + } + if census.RetainedCount != 4 { + t.Fatalf("retainedCount = %d, want 4", census.RetainedCount) + } + if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "mystery" || census.UnknownStatuses[0].Count != 2 { + t.Fatalf("unknownStatuses = %v, want [{mystery 2}]", census.UnknownStatuses) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty (runs has no routine attribution)", census.TopRoutineID) + } +} + +func TestCensusRunEvents_PlainCountNoStatusDetection(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRunEvent(t, conn, runID, 1) + seedRunEvent(t, conn, runID, 2) + + census, err := store.CensusRunEvents(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRunEvents: %v", err) + } + if census.RetainedCount != 2 { + t.Fatalf("retainedCount = %d, want 2", census.RetainedCount) + } + if census.UnknownStatuses != nil { + t.Fatalf("unknownStatuses = %v, want nil", census.UnknownStatuses) + } +} + +// TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent +// proves the fix for the top-routine attribution's former two-query race: a +// write landing between the unknown-status scan and the single-statement +// totals read must not let the reported TopRoutineShare and RetainedCount +// come from different snapshots of the table. Before the fix, this seam sat +// between two independent reads and could make TopRoutineShare exceed 1.0 +// or attribute a share against a stale total. +func TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutine(t, conn, "r-2") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount) + } + if census.TopRoutineID != "r-2" { + t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 2.0/3.0; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } + if census.TopRoutineShare > 1.0 { + t.Fatalf("topRoutineShare = %v, must never exceed 1.0", census.TopRoutineShare) + } +} + +// TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError +// proves routineRunCensusTotals treats a table that became empty as the +// legitimate zero state rather than propagating sql.ErrNoRows: the old +// two-query design decided whether to run the top-routine query from a +// separately-read, now-stale nonzero total, so this same interleaving used +// to surface an unhandled error instead of a clean zero census. +func TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + if _, err := conn.Exec(`DELETE FROM office_routine_runs`); err != nil { + t.Fatalf("delete all rows: %v", err) + } + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 0 { + t.Fatalf("retainedCount = %d, want 0", census.RetainedCount) + } + if census.TopRoutineID != "" { + t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID) + } +} + +func timePtr(t time.Time) *time.Time { return &t } diff --git a/apps/backend/internal/office/retention/store_postgres_test.go b/apps/backend/internal/office/retention/store_postgres_test.go new file mode 100644 index 00000000000..91e870e5786 --- /dev/null +++ b/apps/backend/internal/office/retention/store_postgres_test.go @@ -0,0 +1,184 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + "github.com/kandev/kandev/internal/testutil" +) + +// openSharedSchemaPostgresConn opens a second, independent PostgreSQL +// connection pointed at the same isolated test schema as an existing +// connection. A real second physical connection is required to hold a row +// lock that a concurrent statement on the first connection genuinely blocks +// on — a sequential in-process test hook cannot reproduce that. +func openSharedSchemaPostgresConn(t *testing.T, dsn, schema string) *sqlx.DB { + t.Helper() + raw, err := db.OpenPostgres(dsn, 1, 1) + if err != nil { + t.Fatalf("open second postgres connection: %v", err) + } + conn := sqlx.NewDb(raw, "pgx") + conn.SetMaxOpenConns(1) + conn.SetMaxIdleConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := conn.Exec("SET search_path TO " + schema); err != nil { + t.Fatalf("set search_path on second connection: %v", err) + } + return conn +} + +// TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow +// is the regression test for deleteRunsByIDs' missing direct status/age +// predicate (AC-OFFICE-RUN-HISTORY-RETENTION-002.4): a run resurrected by a +// concurrent transaction that has not yet committed when DeleteRunBatch +// selects it, but commits while the DELETE statement is blocked acquiring +// the row's lock — driving PostgreSQL's real EvalPlanQual recheck path. The +// package's existing resurrection tests +// (TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives) use a +// sequential in-process hook that resurrects strictly before the DELETE +// statement starts; they cannot reach this mid-statement window, which +// needs a second, genuinely concurrent connection. +func TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + var schema string + if err := conn.Get(&schema, `SELECT current_schema()`); err != nil { + t.Fatalf("select current_schema: %v", err) + } + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + holder := openSharedSchemaPostgresConn(t, dsn, schema) + holderTx, err := holder.BeginTx(ctx, nil) + if err != nil { + t.Fatalf("begin holder tx: %v", err) + } + // Resurrecting the row inside an uncommitted transaction takes its row + // lock immediately, but the new values are not visible to conn's own + // snapshot until Commit below: selectEligibleRunIDs' plain read is + // never blocked by an uncommitted writer, so it still sees the row as + // terminal and eligible, exactly the window this test targets. + if _, err := holderTx.ExecContext(ctx, `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = $1`, runID); err != nil { + t.Fatalf("resurrect run inside holder tx: %v", err) + } + + admin := testutil.OpenIsolatedPostgres(t, dsn) // separate schema; pg_stat_activity is instance-wide, not schema-scoped + deleteDone := make(chan RunBatchResult, 1) + deleteErr := make(chan error, 1) + go func() { + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + deleteErr <- err + return + } + deleteDone <- result + }() + + // Wait for DeleteRunBatch's DELETE statement to actually be blocked on + // the holder's row lock before committing, so the recheck this test + // targets is guaranteed to happen rather than racing ahead of it. + deadline := time.Now().Add(10 * time.Second) + for { + var waiting bool + if err := admin.GetContext(ctx, &waiting, ` + SELECT EXISTS( + SELECT 1 FROM pg_stat_activity + WHERE wait_event_type = 'Lock' AND query ILIKE '%DELETE FROM runs%' + )`); err != nil { + t.Fatalf("poll pg_stat_activity: %v", err) + } + if waiting { + break + } + if time.Now().After(deadline) { + t.Fatal("DeleteRunBatch's DELETE never showed up waiting on the holder's row lock") + } + time.Sleep(20 * time.Millisecond) + } + + if err := holderTx.Commit(); err != nil { + t.Fatalf("commit holder tx: %v", err) + } + + var result RunBatchResult + select { + case result = <-deleteDone: + case err := <-deleteErr: + t.Fatalf("DeleteRunBatch: %v", err) + case <-time.After(10 * time.Second): + t.Fatal("DeleteRunBatch did not complete after the holder committed") + } + + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op: the row was resurrected before the delete's row lock was granted, so PostgreSQL's own recheck of the row must see it as no longer eligible", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted despite the concurrent recheck") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted despite the concurrent recheck") + } +} + +// TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite is +// AC-OFFICE-RUN-HISTORY-RETENTION-005.1's mandated backlog-path parity test: +// batch selection order is fixed by named columns +// (completion_time ASC, id ASC) rather than left to the engine, so this +// must select and delete the exact same rows on PostgreSQL as +// TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns proves on +// SQLite for the identical settings and starting rows. +func TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + oldest, middle, newest := newID(), newID(), newID() + oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45) + seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC) + seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC) + seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC) + + // floor 0 so all three are eligible; batch limit 2 -> the two oldest + // go, the newest survives as backlog — same as the SQLite test. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 2 { + t.Fatalf("deleted = %d, want 2", deleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest); n != 1 { + t.Fatal("newest row was deleted on PostgreSQL; oldest-first ordering violated") + } + for _, id := range []string{oldest, middle} { + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 { + t.Fatalf("row %s (older) still present on PostgreSQL after batch limit 2", id) + } + } + + eligible, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if eligible != 1 { + t.Fatalf("remaining eligible = %d, want 1 (backlog: the newest row is still eligible, just not yet batched)", eligible) + } +} diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go new file mode 100644 index 00000000000..3db5aed1918 --- /dev/null +++ b/apps/backend/internal/office/retention/store_test.go @@ -0,0 +1,671 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/google/uuid" + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// testDB builds a fresh in-memory SQLite database carrying the real office +// schema (including the retention indexes), so eligibility queries run +// against the genuine table shapes rather than a hand-rolled fixture. +func testDB(t *testing.T) *sqlx.DB { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + return conn +} + +func seedRoutine(t *testing.T, conn *sqlx.DB, id string) { + t.Helper() + now := time.Now().UTC() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routines (id, workspace_id, name, created_at, updated_at) + VALUES (?, 'ws-1', ?, ?, ?) + `), id, id, now, now); err != nil { + t.Fatalf("seed routine %s: %v", id, err) + } +} + +func seedRoutineRun(t *testing.T, conn *sqlx.DB, id, routineID, status string, completedAt *time.Time, createdAt time.Time) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at) + VALUES (?, ?, 'trigger', ?, ?, ?) + `), id, routineID, status, completedAt, createdAt); err != nil { + t.Fatalf("seed routine run %s: %v", id, err) + } +} + +func seedRoutineRunWithFingerprintAndLinkedTask( + t *testing.T, conn *sqlx.DB, id, routineID, status string, + completedAt *time.Time, createdAt time.Time, fingerprint, linkedTaskID string, +) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at, dispatch_fingerprint, linked_task_id) + VALUES (?, ?, 'trigger', ?, ?, ?, ?, ?) + `), id, routineID, status, completedAt, createdAt, fingerprint, linkedTaskID); err != nil { + t.Fatalf("seed routine run %s: %v", id, err) + } +} + +func seedRun(t *testing.T, conn *sqlx.DB, id, agentProfileID, status string, finishedAt *time.Time, requestedAt time.Time) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO runs (id, agent_profile_id, reason, status, requested_at, finished_at) + VALUES (?, ?, 'test', ?, ?, ?) + `), id, agentProfileID, status, requestedAt, finishedAt); err != nil { + t.Fatalf("seed run %s: %v", id, err) + } +} + +func seedPauseRecovery(t *testing.T, conn *sqlx.DB, agentID, taskID, failedRunID string) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_agent_pause_recoveries (agent_id, task_id, failed_run_id) + VALUES (?, ?, ?) + `), agentID, taskID, failedRunID); err != nil { + t.Fatalf("seed pause recovery for run %s: %v", failedRunID, err) + } +} + +func seedRunEvent(t *testing.T, conn *sqlx.DB, runID string, seq int) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO run_events (run_id, seq, event_type, created_at) + VALUES (?, ?, 'test', ?) + `), runID, seq, time.Now().UTC()); err != nil { + t.Fatalf("seed run event for %s: %v", runID, err) + } +} + +func seedRouteAttempt(t *testing.T, conn *sqlx.DB, runID string, seq int) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_run_route_attempts (run_id, seq, provider_id, model, tier, outcome, started_at) + VALUES (?, ?, 'p', 'm', 't', 'ok', ?) + `), runID, seq, time.Now().UTC()); err != nil { + t.Fatalf("seed route attempt for %s: %v", runID, err) + } +} + +func seedRunSkill(t *testing.T, conn *sqlx.DB, runID, skillID string) { + t.Helper() + if _, err := conn.Exec(conn.Rebind(` + INSERT INTO office_run_skills (run_id, skill_id, version, content_hash, materialized_path) + VALUES (?, ?, 'v1', 'hash', '/path') + `), runID, skillID); err != nil { + t.Fatalf("seed run skill for %s: %v", runID, err) + } +} + +func countRows(t *testing.T, conn *sqlx.DB, query string, args ...any) int64 { + t.Helper() + var n int64 + if err := conn.Get(&n, conn.Rebind(query), args...); err != nil { + t.Fatalf("count query %q: %v", query, err) + } + return n +} + +func daysAgo(n int) time.Time { return time.Now().UTC().AddDate(0, 0, -n) } + +func newID() string { return uuid.New().String() } + +func TestCountEligibleRoutineRuns_RespectsStatusWindowAndFloor(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + // 3 history rows older than the window, floor 50: all inside the floor, + // none eligible. + for i := 0; i < 3; i++ { + completed := daysAgo(40 + i) + seedRoutineRun(t, conn, newID(), routineID, "coalesced", &completed, completed) + } + // A task_created (live-state) row, ancient: never eligible regardless of + // age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1). + ancient := daysAgo(3650) + seedRoutineRun(t, conn, newID(), routineID, "task_created", nil, ancient) + + count, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if count != 0 { + t.Fatalf("count = %d, want 0 (all 3 history rows within the floor of 50)", count) + } + + // Add 50 more, all older than window: total history rows now 53, floor + // 50, so exactly 3 are eligible (the 3 oldest, since floor keeps the + // newest 50 of the 53). + for i := 0; i < 50; i++ { + completed := daysAgo(35 + i) + seedRoutineRun(t, conn, newID(), routineID, "done", &completed, completed) + } + count, err = store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50) + if err != nil { + t.Fatalf("CountEligibleRoutineRuns: %v", err) + } + if count != 3 { + t.Fatalf("count = %d, want 3 (53 history rows, floor 50)", count) + } +} + +func TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + var oldest, middle, newest string + oldest, middle, newest = newID(), newID(), newID() + oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45) + seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC) + seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC) + seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC) + + // floor 0 so all three are eligible; batch limit 2 -> the two oldest go, + // the newest survives as backlog. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 2 { + t.Fatalf("deleted = %d, want 2", deleted) + } + remaining := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest) + if remaining != 1 { + t.Fatalf("newest row was deleted; oldest-first ordering violated") + } + for _, id := range []string{oldest, middle} { + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 { + t.Fatalf("row %s (older) still present after batch limit 2", id) + } + } +} + +func TestDeleteRoutineRunsBatch_FloorReassertedAtDeleteTime(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + routineID := newID() + seedRoutine(t, conn, routineID) + + cutoff := daysAgo(30) + c := daysAgo(90) + id := newID() + seedRoutineRun(t, conn, id, routineID, "done", &c, c) + + // Floor 1 (>= the single row present) means the row is protected: it + // is the newest (and only) row for its routine. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 1, 100) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 0 { + t.Fatalf("deleted = %d, want 0 (row is inside the floor)", deleted) + } +} + +// TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine proves +// AC-OFFICE-RUN-HISTORY-RETENTION-001.4's floor is per-owner: every seed +// helper elsewhere in this suite uses a single routine, so a regression +// that dropped routineRunEligibleSubquery's PARTITION BY routine_id +// (turning a per-owner floor into one shared across every routine) would +// otherwise leave the whole suite green. +func TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + routineA, routineB := newID(), newID() + seedRoutine(t, conn, routineA) + seedRoutine(t, conn, routineB) + + cutoff := daysAgo(30) + var newestA, newestB string + for i := 0; i < 5; i++ { + completed := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86) + idA, idB := newID(), newID() + seedRoutineRun(t, conn, idA, routineA, "done", &completed, completed) + seedRoutineRun(t, conn, idB, routineB, "done", &completed, completed) + if i == 4 { + newestA, newestB = idA, idB + } + } + + // Floor 3 per routine: each routine has 5 history rows, so 2 are + // eligible per routine, 4 total. A floor shared across both routines + // (10 rows, floor 3) would instead delete 7 and could delete either + // routine's newest row. + deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 3, 100) + if err != nil { + t.Fatalf("DeleteRoutineRunsBatch: %v", err) + } + if deleted != 4 { + t.Fatalf("deleted = %d, want 4 (2 eligible per routine, floor 3 held independently)", deleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineA); n != 3 { + t.Fatalf("routineA remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineB); n != 3 { + t.Fatalf("routineB remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestA); n != 1 { + t.Fatalf("routineA's newest row was deleted; its floor should have protected it") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestB); n != 1 { + t.Fatalf("routineB's newest row was deleted; its floor should have protected it") + } +} + +func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + seedRunEvent(t, conn, runID, 1) + seedRouteAttempt(t, conn, runID, 0) + seedRunSkill(t, conn, runID, "skill-1") + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 1 || result.RunEventsDeleted != 2 || result.RouteAttemptsDeleted != 1 || result.RunSkillsDeleted != 1 { + t.Fatalf("result = %+v, want RunsDeleted=1 RunEventsDeleted=2 RouteAttemptsDeleted=1 RunSkillsDeleted=1", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d run_events rows remain referencing a deleted run", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_route_attempts WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d route attempt rows remain referencing a deleted run", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_skills WHERE run_id = ?`, runID); n != 0 { + t.Fatalf("%d run skill rows remain referencing a deleted run", n) + } +} + +// TestRunRetention_PreservesActivePauseRecoveryRun proves that a failed run +// still referenced by MarkAgentPausedFixed remains available until its +// recovery snapshot is consumed or discarded. Count and delete must use the +// same protection predicate so a preview cannot promise deletion that the +// batch path applies. +func TestRunRetention_PreservesActivePauseRecoveryRun(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + failedAt := daysAgo(60) + seedRun(t, conn, runID, "agent-recovery", "failed", &failedAt, failedAt) + seedPauseRecovery(t, conn, "agent-recovery", "task-recovery", runID) + + cutoff := daysAgo(30) + count, err := store.CountEligibleRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRuns: %v", err) + } + if count != 0 { + t.Fatalf("eligible count = %d, want 0 while pause recovery references the run", count) + } + + result, err := store.DeleteRunBatch(ctx, conn, cutoff, 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op while recovery is active", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("recovery run count = %d, want 1", n) + } + + if _, err := conn.Exec(conn.Rebind( + `DELETE FROM office_agent_pause_recoveries WHERE agent_id = ? AND task_id = ?`, + ), "agent-recovery", "task-recovery"); err != nil { + t.Fatalf("discard pause recovery: %v", err) + } + + count, err = store.CountEligibleRuns(ctx, conn, cutoff, 0) + if err != nil { + t.Fatalf("CountEligibleRuns after recovery discard: %v", err) + } + if count != 1 { + t.Fatalf("eligible count after recovery discard = %d, want 1", count) + } + result, err = store.DeleteRunBatch(ctx, conn, cutoff, 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch after recovery discard: %v", err) + } + if result.RunsDeleted != 1 || result.Abandoned { + t.Fatalf("result after recovery discard = %+v, want one deleted run", result) + } +} + +// TestDeleteRunBatch_SurvivingRunKeepsEveryEvent proves +// AC-OFFICE-RUN-HISTORY-RETENTION-001.6: run_events is never deleted for a +// run that is not itself being deleted in the same transaction, no matter +// how old those events are. +func TestDeleteRunBatch_SurvivingRunKeepsEveryEvent(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + liveRunID := newID() + // queued: live state, never eligible at any age. + seedRun(t, conn, liveRunID, "agent-1", "queued", nil, daysAgo(400)) + for i := 0; i < 5; i++ { + seedRunEvent(t, conn, liveRunID, i) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 { + t.Fatalf("RunsDeleted = %d, want 0 (queued run is live state)", result.RunsDeleted) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, liveRunID); n != 5 { + t.Fatalf("run_events for a surviving run = %d, want 5 untouched", n) + } +} + +// TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected proves a +// run resurrected to "queued" (as ScheduleRetry does, clearing finished_at) +// before DeleteRunBatch runs at all is excluded by the eligibility +// selection itself, leaving it and its satellites untouched. This is the +// simple case; the delete-time re-assertion this package's AC-002.4 +// re-assertion actually catches — a resurrection landing between +// selection and the delete statement, inside one attempt — is covered by +// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives below. +func TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + // Simulate the resurrection race by racing a resurrecting UPDATE + // against DeleteRunBatch using a second connection to the same + // in-memory database (SQLite's single-writer serializes them, but the + // delete's re-assertion inside the transaction is what must catch the + // now-live row regardless of interleaving — resurrecting up front is + // the deterministic way to exercise that same code path without + // depending on goroutine scheduling). + if _, err := conn.Exec(conn.Rebind(` + UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ? + `), runID); err != nil { + t.Fatalf("resurrect run: %v", err) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op (resurrected run was never selected)", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted") + } +} + +// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives drives +// the retry path deterministically via testAfterSelectEligibleRunIDs: the +// row is eligible at selection time (so it enters attempt 0's id list), +// resurrected by the hook immediately after that selection (before +// deleteRunBatchOnce's own DELETE runs), so the re-assertion inside the +// transaction detects the mismatch, rolls back, and attempt 1 re-selects +// against the now-live row and finds nothing to do. The run and its +// satellite survive throughout. +func TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + t.Cleanup(func() { testAfterSelectEligibleRunIDs = nil }) + testAfterSelectEligibleRunIDs = func(attempt int, ids []string) { + if attempt != 0 { + return + } + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`, + ), runID); err != nil { + t.Fatalf("resurrect run mid-transaction: %v", err) + } + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 0 || result.Abandoned { + t.Fatalf("result = %+v, want a clean no-op after the retry re-selects nothing", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("resurrected run was deleted despite the retry") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("resurrected run's event was deleted despite the retry") + } +} + +// TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches forces the +// resurrection race to land on *both* attempts via +// testAfterSelectEligibleRunIDs: the row is repeatedly resurrected right +// after each selection, so both attempts' delete re-assertions mismatch. +// DeleteRunBatch must give up rather than loop forever, reporting the +// batch as Abandoned — which the sweep records as that table's failure, +// not backlog (AC-OFFICE-RUN-HISTORY-RETENTION-002.7). +func TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + runID := newID() + finished := daysAgo(60) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, runID, 0) + + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + beforeCalls, afterCalls := 0, 0 + // Before each attempt's selection, make sure the row reads terminal + // again (undoing the previous attempt's resurrection) so it is + // selected every time, not just on attempt 0. + testBeforeSelectEligibleRunIDs = func(attempt int) { + beforeCalls++ + if attempt == 0 { + return // already terminal from seeding + } + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`, + ), finished, runID); err != nil { + t.Fatalf("re-terminalize run before attempt %d: %v", attempt, err) + } + } + // After each attempt's selection (which just proved the row was + // terminal), flip it live so that attempt's own delete re-assertion + // mismatches. + testAfterSelectEligibleRunIDs = func(attempt int, ids []string) { + afterCalls++ + if _, err := conn.Exec(conn.Rebind( + `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`, + ), runID); err != nil { + t.Fatalf("resurrect run mid-transaction (attempt %d): %v", attempt, err) + } + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if !result.Abandoned { + t.Fatalf("result = %+v, want Abandoned=true after two consecutive mismatches", result) + } + if result.RunsDeleted != 0 || result.RunEventsDeleted != 0 { + t.Fatalf("result = %+v, want every count zero on an abandoned batch", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 { + t.Fatalf("run was deleted despite an abandoned batch") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 { + t.Fatalf("run_events was deleted despite an abandoned batch") + } + if beforeCalls != 2 || afterCalls != 2 { + t.Fatalf("before/after hooks invoked %d/%d times, want exactly 2/2 (one per attempt)", beforeCalls, afterCalls) + } +} + +// TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile is the runs-table +// equivalent of TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine: +// every other DeleteRunBatch test in this file uses a single +// "agent-1" owner, so a regression that dropped runEligibleSubquery's +// PARTITION BY agent_profile_id would otherwise leave the whole suite green. +func TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + cutoff := daysAgo(30) + var newestA, newestB string + for i := 0; i < 5; i++ { + finished := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86) + idA, idB := newID(), newID() + seedRun(t, conn, idA, "agent-a", "finished", &finished, finished) + seedRun(t, conn, idB, "agent-b", "finished", &finished, finished) + if i == 4 { + newestA, newestB = idA, idB + } + } + + // Floor 3 per agent profile: each owns 5 history rows, so 2 are + // eligible per owner, 4 total. + result, err := store.DeleteRunBatch(ctx, conn, cutoff, 3, 100) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.RunsDeleted != 4 || result.Abandoned { + t.Fatalf("result = %+v, want 4 deleted (2 eligible per agent profile, floor 3 held independently)", result) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-a"); n != 3 { + t.Fatalf("agent-a remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-b"); n != 3 { + t.Fatalf("agent-b remaining = %d, want 3 (its own floor)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestA); n != 1 { + t.Fatalf("agent-a's newest run was deleted; its floor should have protected it") + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestB); n != 1 { + t.Fatalf("agent-b's newest run was deleted; its floor should have protected it") + } +} + +// TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements proves +// deleteRunBatchOnce's satellite and runs deletes split their id list into +// chunks of at most retentionMaxHostParams rather than binding the whole +// batch as one IN clause, which is what let batch_limit's documented range +// (AC-OFFICE-RUN-HISTORY-RETENTION-004.3, up to 100,000) overflow a single +// statement's bind-parameter limit on either engine. retentionMaxHostParams+2 +// runs, each with one satellite row apiece, forces the delete loop to span +// more than one chunk; every row and every satellite must still be deleted +// and the reported count must reflect the true total, not just one chunk's. +func TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + + const rowCount = retentionMaxHostParams + 2 + finished := daysAgo(60) + ids := make([]string, 0, rowCount) + for i := 0; i < rowCount; i++ { + id := newID() + seedRun(t, conn, id, "agent-1", "finished", &finished, finished) + seedRunEvent(t, conn, id, 0) + ids = append(ids, id) + } + + result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, rowCount) + if err != nil { + t.Fatalf("DeleteRunBatch: %v", err) + } + if result.Abandoned { + t.Fatalf("result = %+v, want a clean delete, not abandoned", result) + } + if result.RunsDeleted != int64(rowCount) { + t.Fatalf("RunsDeleted = %d, want %d", result.RunsDeleted, rowCount) + } + if result.RunEventsDeleted != int64(rowCount) { + t.Fatalf("RunEventsDeleted = %d, want %d", result.RunEventsDeleted, rowCount) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 { + t.Fatalf("runs remaining = %d, want 0 (every row across every chunk must be deleted)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events`); n != 0 { + t.Fatalf("run_events remaining = %d, want 0", n) + } + for _, id := range ids { + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, id); n != 0 { + t.Fatalf("run %s remains after a chunked delete", id) + } + } +} + +func TestCountRunEvents_PlainCount(t *testing.T) { + conn := testDB(t) + store := NewStore(db.NewPool(conn, conn)) + ctx := context.Background() + runID := newID() + finished := daysAgo(1) + seedRun(t, conn, runID, "agent-1", "finished", &finished, finished) + for i := 0; i < 4; i++ { + seedRunEvent(t, conn, runID, i) + } + count, err := store.CountRunEvents(ctx, conn) + if err != nil { + t.Fatalf("CountRunEvents: %v", err) + } + if count != 4 { + t.Fatalf("CountRunEvents() = %d, want 4", count) + } +} diff --git a/apps/backend/internal/office/retention/sweep.go b/apps/backend/internal/office/retention/sweep.go new file mode 100644 index 00000000000..9c972d19cf4 --- /dev/null +++ b/apps/backend/internal/office/retention/sweep.go @@ -0,0 +1,331 @@ +package retention + +import ( + "context" + "sync" + "time" + + "github.com/kandev/kandev/internal/db" +) + +// TableSweepResult is one reported table's outcome for one sweep +// (AC-OFFICE-RUN-HISTORY-RETENTION-004.6). Backlog is only ever true for a +// swept table's own SweptTableResult; a satellite table's Backlog stays +// false because its deletion is never independently batch-limited — it +// deletes exactly the run ids its owning runs batch selected. +type TableSweepResult struct { + Deleted int64 `json:"deleted"` + Backlog bool `json:"backlog"` + Err string `json:"error"` +} + +// SweptTableResult adds preview reporting to a swept table's outcome. +// Previewed is true only when THIS sweep was a preview pass for this +// table, never a running total. +type SweptTableResult struct { + TableSweepResult + Previewed bool `json:"previewed"` + WouldDelete int64 `json:"would_delete"` +} + +// LastSweep is the in-memory value replaced wholesale at the end of each +// sweep that actually ran (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7). +// Nothing here is persisted. +type LastSweep struct { + StartedAt time.Time `json:"started_at"` + FinishedAt time.Time `json:"finished_at"` + + OfficeRoutineRuns SweptTableResult `json:"office_routine_runs"` + Runs SweptTableResult `json:"runs"` + RunEvents TableSweepResult `json:"run_events"` + RouteAttempts TableSweepResult `json:"route_attempts"` + RunSkills TableSweepResult `json:"run_skills"` +} + +type satelliteResults struct { + RunEvents TableSweepResult + RouteAttempts TableSweepResult + RunSkills TableSweepResult +} + +// Sweeper runs one sweep or one census pass at a time; the scheduler +// (scheduler.go) owns when to call each. +// +// - Lost-exclusivity outcome: a lock lost between tables abandons the +// whole sweep attempt as a skip — the same outcome as a local +// concurrent-sweep collision — rather than inventing a per-table +// "skipped" state the design's eight-id health catalogue has nowhere +// to report. Batches already committed on PostgreSQL stay committed +// (AC-002.5); LastSweep simply is not replaced by this attempt, and +// the next scheduled sweep reports the tables' true state either way. +// - The skip counter increments for a collision, a failed settings +// re-read, a failed lock acquisition, and a lock lost mid-sweep — every +// case where a sweep was due and did not produce a result. It does +// NOT increment when retention is disabled (AC-002.8's "run no sweep" +// is a deliberate, steady-state condition, not a due sweep that could +// not run; incrementing forever while off would make the counter +// meaningless). +type Sweeper struct { + pool *db.Pool + store *Store + settingsStore *SettingsStore + previewMarker *PreviewMarkerStore + census *CensusTracker + + mu sync.Mutex + sweeping bool + lastSweep *LastSweep + skipCount int64 + lastSkip time.Time +} + +// NewSweeper wires the sweep orchestration to its dependencies. +func NewSweeper(pool *db.Pool, store *Store, settingsStore *SettingsStore, previewMarker *PreviewMarkerStore) *Sweeper { + return &Sweeper{ + pool: pool, + store: store, + settingsStore: settingsStore, + previewMarker: previewMarker, + census: NewCensusTracker(), + } +} + +// LastSweepSnapshot returns the most recent completed sweep's result. ok is +// false before the first sweep ever completes (AC-OFFICE-RUN-HISTORY-RETENTION-004.7). +func (s *Sweeper) LastSweepSnapshot() (LastSweep, bool) { + s.mu.Lock() + defer s.mu.Unlock() + if s.lastSweep == nil { + return LastSweep{}, false + } + return *s.lastSweep, true +} + +// SkipSnapshot returns the running skip count and the last skip's time. +func (s *Sweeper) SkipSnapshot() (count int64, lastAt time.Time) { + s.mu.Lock() + defer s.mu.Unlock() + return s.skipCount, s.lastSkip +} + +// CensusSnapshot returns the current per-table retained counts. +func (s *Sweeper) CensusSnapshot() RetainedCounts { + return s.census.Snapshot() +} + +// RunSweep performs at most one sweep: office_routine_runs, then runs with +// its satellites, in that fixed order (AC-OFFICE-RUN-HISTORY-RETENTION-002.6). +func (s *Sweeper) RunSweep(ctx context.Context) { + if !s.beginSweep() { + s.recordSkip() + return + } + defer s.endSweep() + + settings, err := s.settingsStore.GetSettingsForSweep(ctx) + if err != nil { + // Settings unreadable at sweep start fails closed: no sweep, no + // fallback to the (possibly shorter) default window. + s.recordSkip() + return + } + if !settings.Enabled { + // AC-002.8: disabled means no sweep at all, not a recorded skip. + return + } + + q, session, ok := s.acquireQueryer(ctx) + if !ok { + s.recordSkip() + return + } + if session != nil { + defer session.release() + } + + now := time.Now().UTC() + report := LastSweep{StartedAt: now} + report.OfficeRoutineRuns = s.sweepRoutineRuns(ctx, q, settings.RoutineRuns, now, settings.BatchLimit) + + if testBetweenTablesSweep != nil { + testBetweenTablesSweep(q) + } + if session != nil && !session.alive(ctx) { + // Exclusivity was lost after the first table's work. Do not + // start the second table, and do not publish a partial result — + // this whole attempt is a skip, exactly as if the lock had + // never been acquired. + s.recordSkip() + return + } + + runsResult, satellites := s.sweepRuns(ctx, q, settings.Runs, now, settings.BatchLimit) + report.Runs = runsResult + report.RunEvents = satellites.RunEvents + report.RouteAttempts = satellites.RouteAttempts + report.RunSkills = satellites.RunSkills + + report.FinishedAt = time.Now().UTC() + s.mu.Lock() + s.lastSweep = &report + s.mu.Unlock() + + incSweepCompleted() + incDeleted(TableOfficeRoutineRuns, report.OfficeRoutineRuns.Deleted) + incDeleted(TableRuns, report.Runs.Deleted) + incDeleted(TableRunEvents, report.RunEvents.Deleted) + incDeleted("office_run_route_attempts", report.RouteAttempts.Deleted) + incDeleted("office_run_skills", report.RunSkills.Deleted) +} + +// testBetweenTablesSweep, when set, runs right after office_routine_runs' +// table work and right before the alive() liveness check and the runs +// table — a deterministic seam for exercising AC-OFFICE-RUN-HISTORY-RETENTION-002.12's +// "verifies the lock connection is still alive between tables" path +// without depending on real cross-process timing. It receives the sweep's +// own queryer so a test can run diagnostics (or a second sweep attempt) +// against the exact connection in use. Never set outside tests. +var testBetweenTablesSweep func(q queryer) + +// RunCensus evaluates the retained-count census for every thresholded +// table. Read-only, so it needs no advisory lock: every backend computes +// and serves its own local view. +func (s *Sweeper) RunCensus(ctx context.Context) { + q := s.pool.Reader() + now := time.Now().UTC() + + routineCensus, err := s.store.CensusRoutineRuns(ctx, q, now) + s.census.RecordRoutineRuns(routineCensus, err) + incCensus(TableOfficeRoutineRuns, err) + + runsCensus, err := s.store.CensusRuns(ctx, q, now) + s.census.RecordRuns(runsCensus, err) + incCensus(TableRuns, err) + + runEventsCensus, err := s.store.CensusRunEvents(ctx, q, now) + s.census.RecordRunEvents(runEventsCensus, err) + incCensus(TableRunEvents, err) +} + +func (s *Sweeper) beginSweep() bool { + s.mu.Lock() + defer s.mu.Unlock() + if s.sweeping { + return false + } + s.sweeping = true + return true +} + +func (s *Sweeper) endSweep() { + s.mu.Lock() + s.sweeping = false + s.mu.Unlock() +} + +func (s *Sweeper) recordSkip() { + s.mu.Lock() + defer s.mu.Unlock() + s.skipCount++ + s.lastSkip = time.Now().UTC() + incSweepSkipped() +} + +func (s *Sweeper) acquireQueryer(ctx context.Context) (queryer, *sweepSession, bool) { + if !s.store.IsPostgres() { + return s.pool.Writer(), nil, true + } + session, ok, err := acquireSweepSession(ctx, s.pool) + if err != nil || !ok { + return nil, nil, false + } + return session.queryer(), session, true +} + +// isPreviewed reads the preview marker through q, the sweep's own +// connection, rather than the shared settings pool (see the +// PreviewMarkerStore.GetWith doc comment). +func (s *Sweeper) isPreviewed(ctx context.Context, q queryer, table TableName) bool { + marker, _ := s.previewMarker.GetWith(ctx, q) + _, ok := marker[table] + return ok +} + +func (s *Sweeper) sweepRoutineRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) SweptTableResult { + cutoff := retentionCutoff(now, cfg.WindowDays) + if !s.isPreviewed(ctx, q, TableOfficeRoutineRuns) { + return s.previewTable(ctx, q, TableOfficeRoutineRuns, func() (int64, error) { + return s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner) + }, now) + } + + eligible, err := s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + deleted, err := s.store.DeleteRoutineRunsBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + return SweptTableResult{TableSweepResult: TableSweepResult{ + Deleted: deleted, + Backlog: eligible > int64(batchLimit), + }} +} + +func (s *Sweeper) sweepRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) (SweptTableResult, satelliteResults) { + cutoff := retentionCutoff(now, cfg.WindowDays) + if !s.isPreviewed(ctx, q, TableRuns) { + result := s.previewTable(ctx, q, TableRuns, func() (int64, error) { + return s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner) + }, now) + return result, satelliteResults{} + } + + eligible, err := s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} + } + result, err := s.store.DeleteRunBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit) + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{} + } + if result.Abandoned { + // AC-002.7: an abandoned batch is that table's failure, not + // backlog; the satellites report zero, matching what the + // rollback actually left committed. + return SweptTableResult{TableSweepResult: TableSweepResult{ + Err: "batch abandoned after two consecutive mismatches", + }}, satelliteResults{} + } + return SweptTableResult{TableSweepResult: TableSweepResult{ + Deleted: result.RunsDeleted, + Backlog: eligible > int64(batchLimit), + }}, satelliteResults{ + RunEvents: TableSweepResult{Deleted: result.RunEventsDeleted}, + RouteAttempts: TableSweepResult{Deleted: result.RouteAttemptsDeleted}, + RunSkills: TableSweepResult{Deleted: result.RunSkillsDeleted}, + } +} + +// retentionCutoff turns a table's configured window into the instant a +// history row's completion time must be older than to be eligible, +// derived from the one sweep-start instant both tables share +// (AC-OFFICE-RUN-HISTORY-RETENTION-002.11). +func retentionCutoff(now time.Time, windowDays int) time.Time { + return now.AddDate(0, 0, -windowDays) +} + +// previewTable runs a table's first-ever preview pass: count eligible rows +// uncapped, mark the table previewed, and delete nothing +// (AC-OFFICE-RUN-HISTORY-RETENTION-003.2, -003.9). +func (s *Sweeper) previewTable(ctx context.Context, q queryer, table TableName, countEligible func() (int64, error), now time.Time) SweptTableResult { + count, err := countEligible() + if err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + if err := s.previewMarker.MarkCompletedWith(ctx, q, table, now); err != nil { + return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}} + } + return SweptTableResult{Previewed: true, WouldDelete: count} +} diff --git a/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go new file mode 100644 index 00000000000..7c9ed78b9a6 --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go @@ -0,0 +1,113 @@ +package retention + +import ( + "context" + "testing" + + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" +) + +// TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep proves +// AC-OFFICE-RUN-HISTORY-RETENTION-005.3: after a sweep deletes unrelated +// history rows, including ones sharing the same routine and fingerprint, +// the concurrency gate's fingerprint lookup still returns exactly the live +// task_created run it would have returned had no sweep run. +func TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + repo, err := officesqlite.NewWithDB(conn, conn, nil) + if err != nil { + t.Fatalf("open repo: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "done", &old, old, "fp-1", "") + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "failed", &old, old, "fp-1", "") + + activeID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, activeID, "r-1", "task_created", nil, daysAgo(1), "fp-1", "") + + before, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1") + if err != nil { + t.Fatalf("GetActiveRunForFingerprint (before): %v", err) + } + if before == nil || before.ID != activeID { + t.Fatalf("GetActiveRunForFingerprint (before) = %+v, want id %s", before, activeID) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE status IN ('done', 'failed')`); n != 0 { + t.Fatalf("old rows remaining = %d, want 0 (sweep should have deleted them)", n) + } + + after, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1") + if err != nil { + t.Fatalf("GetActiveRunForFingerprint (after): %v", err) + } + if after == nil || after.ID != activeID { + t.Fatalf("GetActiveRunForFingerprint (after sweep) = %+v, want the same active run %s", after, activeID) + } +} + +// TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing proves +// AC-OFFICE-RUN-HISTORY-RETENTION-005.4: once a sweep deletes the routine +// run linked to a task, the task-closure lookup for that task ID resolves +// to nothing rather than to an unrelated run. +func TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + repo, err := officesqlite.NewWithDB(conn, conn, nil) + if err != nil { + t.Fatalf("open repo: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + deletedID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, deletedID, "r-1", "done", &old, old, "fp-1", "task-1") + + // A second, unrelated run under a different task must not be + // mistaken for task-1's closed-out run. + other := daysAgo(1) + otherID := newID() + seedRoutineRunWithFingerprintAndLinkedTask(t, conn, otherID, "r-1", "done", &other, other, "fp-2", "task-2") + + before, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (before): %v", err) + } + if before == nil || before.ID != deletedID { + t.Fatalf("GetRoutineRunByLinkedTaskID (before) = %+v, want id %s", before, deletedID) + } + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, deletedID); n != 0 { + t.Fatalf("deleted run rows = %d, want 0", n) + } + + after, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (after): %v", err) + } + if after != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (after sweep) = %+v, want nil", after) + } + + // task-2's own run must be unaffected by task-1's deletion. + stillThere, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-2") + if err != nil { + t.Fatalf("GetRoutineRunByLinkedTaskID (task-2): %v", err) + } + if stillThere == nil || stillThere.ID != otherID { + t.Fatalf("GetRoutineRunByLinkedTaskID (task-2) = %+v, want id %s", stillThere, otherID) + } +} diff --git a/apps/backend/internal/office/retention/sweep_postgres_test.go b/apps/backend/internal/office/retention/sweep_postgres_test.go new file mode 100644 index 00000000000..06d2d62a7ff --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_postgres_test.go @@ -0,0 +1,238 @@ +package retention + +import ( + "context" + "testing" + "time" + + "github.com/jmoiron/sqlx" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" + taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite" + "github.com/kandev/kandev/internal/testutil" +) + +// newPostgresTestSweeper is sweep_test.go's newTestSweeper, built against a +// real, isolated-schema PostgreSQL connection instead of in-memory SQLite. +// tasks is created first, mirroring production boot order (see +// child_summaries_postgres_test.go). +func newPostgresTestSweeper(t *testing.T, dsn string) (*Sweeper, *sqlx.DB) { + t.Helper() + conn := testutil.OpenIsolatedPostgres(t, dsn) + if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init task repo: %v", err) + } + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + store := NewStore(pool) + settingsStore := NewSettingsStore(settingsRaw) + previewMarker := NewPreviewMarkerStore(settingsRaw) + return NewSweeper(pool, store, settingsStore, previewMarker), conn +} + +// TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables is +// AC-OFFICE-RUN-HISTORY-RETENTION-002.12's mandated two-backend test: the +// seed gives the winner two tables' worth of work, and the loser's attempt +// happens in the pause between them (via testBetweenTablesSweep). A +// transaction-scoped lock would have released between the winner's two +// per-table statements and let the loser in; this must not happen with the +// session-scoped lock. +func TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + winner, conn := newPostgresTestSweeper(t, dsn) + saveZeroFloorSettings(t, winner) + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + winner.RunSweep(ctx) // preview pass for both tables; no lock contention to test yet + + loser, _ := newPostgresTestSweeper(t, dsn) + + testBetweenTablesSweep = func(queryer) { + loser.RunSweep(ctx) + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + winner.RunSweep(ctx) // deleting pass: office_routine_runs, pause, then runs + + winnerLast, ok := winner.LastSweepSnapshot() + if !ok { + t.Fatal("winner LastSweepSnapshot: ok = false, want true") + } + if winnerLast.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("winner office_routine_runs.Deleted = %d, want 1", winnerLast.OfficeRoutineRuns.Deleted) + } + if winnerLast.Runs.Deleted != 1 { + t.Fatalf("winner runs.Deleted = %d, want 1 (winner must complete both tables)", winnerLast.Runs.Deleted) + } + + if _, ok := loser.LastSweepSnapshot(); ok { + t.Fatal("loser LastSweepSnapshot: ok = true, want false (loser must not have run)") + } + loserSkips, _ := loser.SkipSnapshot() + if loserSkips != 1 { + t.Fatalf("loser skip count = %d, want 1", loserSkips) + } +} + +// TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire is +// the companion test the design's Testing section requires: dropping the +// winner's lock connection mid-sweep must stop it before the next table +// (F25) and record the whole attempt as a skip rather than a partial +// result (F28), never attempting to re-acquire. +func TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + victim, conn := newPostgresTestSweeper(t, dsn) + saveZeroFloorSettings(t, victim) + + // The pool's one connection (SetMaxOpenConns(1)) is what pg_terminate_backend + // kills below; database/sql transparently opens a replacement on the + // next query, which starts on the default search_path rather than + // this test's isolated schema. Capture the schema now to restore it + // before any post-mortem query on conn. + var schema string + if err := conn.Get(&schema, `SELECT current_schema()`); err != nil { + t.Fatalf("select current_schema: %v", err) + } + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + victim.RunSweep(ctx) // preview pass for both tables + + admin := testutil.OpenIsolatedPostgres(t, dsn) + adminPool := db.NewPool(admin, admin) + + testBetweenTablesSweep = func(q queryer) { + var pid int + if err := q.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil { + t.Fatalf("select pg_backend_pid: %v", err) + } + if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil { + t.Fatalf("terminate lock connection: %v", err) + } + // pg_terminate_backend signals the backend asynchronously; wait + // for it to actually leave pg_stat_activity before returning, so + // the alive() check right after this hook is not racing the + // signal's delivery. + deadline := time.Now().Add(5 * time.Second) + for { + var stillThere bool + if err := adminPool.Writer().GetContext(ctx, &stillThere, + `SELECT EXISTS(SELECT 1 FROM pg_stat_activity WHERE pid = $1)`, pid, + ); err != nil { + t.Fatalf("poll pg_stat_activity: %v", err) + } + if !stillThere { + return + } + if time.Now().After(deadline) { + t.Fatalf("backend %d still present in pg_stat_activity after 5s", pid) + } + time.Sleep(10 * time.Millisecond) + } + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + beforeAttempt, ok := victim.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot after the preview pass: ok = false, want true") + } + + victim.RunSweep(ctx) // deleting pass: office_routine_runs succeeds, then the lock connection dies + + // The preview pass already set LastSweep; a lock lost mid-attempt + // must leave it exactly as-is rather than publishing a partial + // result (F28) — not become unset, which would also be true after a + // genuinely successful sweep with nothing yet recorded. + afterAttempt, ok := victim.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot after the lost-lock attempt: ok = false, want true (still the preview pass's result)") + } + if afterAttempt != beforeAttempt { + t.Fatalf("LastSweep changed after a lost-lock attempt: before=%+v after=%+v", beforeAttempt, afterAttempt) + } + skips, _ := victim.SkipSnapshot() + if skips != 1 { + t.Fatalf("skip count = %d, want 1", skips) + } + + // conn's one physical connection was the one just terminated; + // database/sql opened a replacement on the default search_path, so + // restore the isolated schema before verifying table state. + if _, err := conn.Exec("SET search_path TO " + schema); err != nil { + t.Fatalf("restore search_path: %v", err) + } + + // office_routine_runs' delete committed before the session died + // (AC-002.5: batches already committed stay committed). + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 { + t.Fatalf("office_routine_runs rows = %d, want 0 (the first table's committed delete survives)", n) + } + // runs was never reached. + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 { + t.Fatalf("runs rows = %d, want 1 (the second table must not have been touched)", n) + } + + // A fresh session can now acquire the lock: it was not re-acquired by + // the victim, and PostgreSQL released it when the session ended. + replacement, ok, err := acquireSweepSession(ctx, adminPool) + if err != nil { + t.Fatalf("acquireSweepSession: %v", err) + } + if !ok { + t.Fatal("ok = false, want true (the terminated session must have released the lock)") + } + replacement.release() +} + +// TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite +// proves routineRunCensusTotals' window-function query — SUM(COUNT(*)) OVER +// () over a GROUP BY routine_id — is valid PostgreSQL and, being one +// statement, cannot be split by a write landing between the unknown-status +// scan and the totals read, unlike the two independent queries it replaced. +func TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite(t *testing.T) { + dsn := testutil.PostgresDSNFromEnv(t) + ctx := context.Background() + + sweeper, conn := newPostgresTestSweeper(t, dsn) + store := sweeper.store + + seedRoutine(t, conn, "r-1") + seedRoutine(t, conn, "r-2") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + + testBetweenRoutineRunCensusReads = func(queryer) { + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1)) + } + t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil }) + + census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC()) + if err != nil { + t.Fatalf("CensusRoutineRuns: %v", err) + } + if census.RetainedCount != 3 { + t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount) + } + if census.TopRoutineID != "r-2" { + t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID) + } + if got, want := census.TopRoutineShare, 2.0/3.0; got != want { + t.Fatalf("topRoutineShare = %v, want %v", got, want) + } +} diff --git a/apps/backend/internal/office/retention/sweep_test.go b/apps/backend/internal/office/retention/sweep_test.go new file mode 100644 index 00000000000..2f4005f11a8 --- /dev/null +++ b/apps/backend/internal/office/retention/sweep_test.go @@ -0,0 +1,453 @@ +package retention + +import ( + "context" + "testing" + + "github.com/jmoiron/sqlx" + _ "github.com/mattn/go-sqlite3" + + "github.com/kandev/kandev/internal/db" + officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite" + systemsettings "github.com/kandev/kandev/internal/system/settings" +) + +// newTestSweeper builds a Sweeper over one in-memory SQLite database +// carrying both the office schema (routine runs, plain runs, satellites) +// and the settings schema, so a sweep's table deletes and its +// settings/preview reads share one connection exactly as they do on the +// writer pool in production. +func newTestSweeper(t *testing.T) (*Sweeper, *sqlx.DB) { + t.Helper() + conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + conn.SetMaxOpenConns(1) + t.Cleanup(func() { _ = conn.Close() }) + if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil { + t.Fatalf("init office schema: %v", err) + } + pool := db.NewPool(conn, conn) + settingsRaw, err := systemsettings.NewStore(pool) + if err != nil { + t.Fatalf("init settings schema: %v", err) + } + + store := NewStore(pool) + settingsStore := NewSettingsStore(settingsRaw) + previewMarker := NewPreviewMarkerStore(settingsRaw) + return NewSweeper(pool, store, settingsStore, previewMarker), conn +} + +func TestRunSweep_FirstPassPreviewsBothTablesWithoutDeleting(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if !last.OfficeRoutineRuns.Previewed || last.OfficeRoutineRuns.WouldDelete != 1 { + t.Fatalf("office_routine_runs = %+v, want previewed with WouldDelete=1", last.OfficeRoutineRuns) + } + if last.OfficeRoutineRuns.Deleted != 0 { + t.Fatalf("office_routine_runs.Deleted = %d, want 0 on a preview pass", last.OfficeRoutineRuns.Deleted) + } + if !last.Runs.Previewed || last.Runs.WouldDelete != 1 { + t.Fatalf("runs = %+v, want previewed with WouldDelete=1", last.Runs) + } + if last.Runs.Deleted != 0 { + t.Fatalf("runs.Deleted = %d, want 0 on a preview pass", last.Runs.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 { + t.Fatalf("office_routine_runs rows after preview = %d, want 1 (nothing deleted)", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 { + t.Fatalf("runs rows after preview = %d, want 1 (nothing deleted)", n) + } +} + +func TestRunSweep_SecondPassDeletesAfterPreview(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + sweeper.RunSweep(ctx) // deleting pass + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if last.OfficeRoutineRuns.Previewed { + t.Fatal("office_routine_runs: second sweep should not be a preview pass") + } + if last.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1", last.OfficeRoutineRuns.Deleted) + } + if last.Runs.Previewed { + t.Fatal("runs: second sweep should not be a preview pass") + } + if last.Runs.Deleted != 1 { + t.Fatalf("runs.Deleted = %d, want 1", last.Runs.Deleted) + } + if last.RunEvents.Deleted != 1 { + t.Fatalf("run_events.Deleted = %d, want 1", last.RunEvents.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 { + t.Fatalf("office_routine_runs rows after delete = %d, want 0", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 { + t.Fatalf("runs rows after delete = %d, want 0", n) + } +} + +// TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow seeds one row +// inside the configured window and one past it, per table, with the floor +// dropped to 0 so only age decides eligibility. A cutoff that ignores +// WindowDays (using the sweep instant itself) would preview and then delete +// both rows instead of only the old one. +func TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) // DefaultSettings() keeps the 30-day window + + seedRoutine(t, conn, "r-1") + recent := daysAgo(1) + old := daysAgo(60) + + recentRoutineRunID := newID() + oldRoutineRunID := newID() + seedRoutineRun(t, conn, recentRoutineRunID, "r-1", "done", &recent, recent) + seedRoutineRun(t, conn, oldRoutineRunID, "r-1", "done", &old, old) + + recentRunID := newID() + oldRunID := newID() + seedRun(t, conn, recentRunID, "agent-1", "finished", &recent, recent) + seedRun(t, conn, oldRunID, "agent-1", "finished", &old, old) + + sweeper.RunSweep(ctx) // preview pass + + preview, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if preview.OfficeRoutineRuns.WouldDelete != 1 { + t.Fatalf("office_routine_runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.OfficeRoutineRuns.WouldDelete) + } + if preview.Runs.WouldDelete != 1 { + t.Fatalf("runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.Runs.WouldDelete) + } + + sweeper.RunSweep(ctx) // deleting pass + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if last.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.OfficeRoutineRuns.Deleted) + } + if last.Runs.Deleted != 1 { + t.Fatalf("runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.Runs.Deleted) + } + + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, recentRoutineRunID); n != 1 { + t.Fatalf("recent routine-run rows = %d, want 1: a row inside the retention window must survive", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, oldRoutineRunID); n != 0 { + t.Fatalf("old routine-run rows = %d, want 0", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, recentRunID); n != 1 { + t.Fatalf("recent run rows = %d, want 1: a row inside the retention window must survive", n) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, oldRunID); n != 0 { + t.Fatalf("old run rows = %d, want 0", n) + } +} + +func TestRunSweep_BacklogFlaggedWhenEligibleExceedsBatchLimit(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + const eligibleRows = 105 + const batchLimit = 100 // minBatchLimit; AC-004.3 forbids going lower + + seedRoutine(t, conn, "r-1") + for i := 0; i < eligibleRows; i++ { + old := daysAgo(60 + i) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + } + + settings := DefaultSettings() + settings.BatchLimit = batchLimit + settings.RoutineRuns.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) // preview pass, no deletion, no batch limit involved + sweeper.RunSweep(ctx) // deleting pass: 105 eligible, batch limit 100 + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false") + } + if last.OfficeRoutineRuns.Deleted != batchLimit { + t.Fatalf("Deleted = %d, want %d (capped by batch limit)", last.OfficeRoutineRuns.Deleted, batchLimit) + } + if !last.OfficeRoutineRuns.Backlog { + t.Fatal("Backlog = false, want true (105 eligible > batch limit 100)") + } +} + +func TestRunSweep_DisabledSkipsSweepWithoutRecordingSkip(t *testing.T) { + sweeper, _ := newTestSweeper(t) + ctx := context.Background() + + settings := DefaultSettings() + settings.Enabled = false + if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false (disabled means no sweep ran)") + } + count, _ := sweeper.SkipSnapshot() + if count != 0 { + t.Fatalf("skip count = %d, want 0 (disabled is not a recorded skip)", count) + } +} + +func TestRunSweep_SettingsUnreadableSkipsAndRecordsSkip(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + if _, err := conn.Exec(` + INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP) + `); err != nil { + t.Fatalf("seed unparseable settings: %v", err) + } + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false") + } + count, lastAt := sweeper.SkipSnapshot() + if count != 1 { + t.Fatalf("skip count = %d, want 1", count) + } + if lastAt.IsZero() { + t.Fatal("lastAt is zero, want a recorded skip time") + } +} + +func TestRunSweep_ConcurrentAttemptRecordsSkipWithoutRunning(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + + // Simulate a sweep already in flight rather than racing goroutines + // against SQLite's speed, which would make the collision + // non-deterministic. + sweeper.mu.Lock() + sweeper.sweeping = true + sweeper.mu.Unlock() + + sweeper.RunSweep(ctx) + + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("LastSweepSnapshot: ok = true, want false (the attempt must not have run)") + } + count, _ := sweeper.SkipSnapshot() + if count != 1 { + t.Fatalf("skip count = %d, want 1", count) + } + if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 { + t.Fatalf("rows = %d, want 1 (a blocked attempt must not preview or delete)", n) + } +} + +func TestRunSweep_AbandonedRunsBatchReportsFailureNotBacklogWithZeroSatellites(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + old := daysAgo(60) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", &old, old) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunSweep(ctx) // preview pass + + // Before each attempt's selection, make sure the row reads terminal + // again (undoing the previous attempt's resurrection) so it is + // selected every time; after selection, flip it live so that + // attempt's own delete re-assertion mismatches — forcing both + // attempts to roll back and the batch to abandon. + testBeforeSelectEligibleRunIDs = func(attempt int) { + if attempt == 0 { + return // already terminal from seeding + } + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID) + } + testAfterSelectEligibleRunIDs = func(int, []string) { + conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID) + } + t.Cleanup(func() { + testBeforeSelectEligibleRunIDs = nil + testAfterSelectEligibleRunIDs = nil + }) + + sweeper.RunSweep(ctx) // deleting pass: forced to abandon + + last, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false") + } + if last.Runs.Err == "" { + t.Fatal("runs.Err is empty, want the abandon failure recorded") + } + if last.Runs.Backlog { + t.Fatal("runs.Backlog = true, want false (an abandoned batch is a failure, not backlog)") + } + if last.Runs.Deleted != 0 { + t.Fatalf("runs.Deleted = %d, want 0", last.Runs.Deleted) + } + if last.RunEvents.Deleted != 0 { + t.Fatalf("run_events.Deleted = %d, want 0 (rollback restored it)", last.RunEvents.Deleted) + } +} + +// TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState +// is AC-OFFICE-RUN-HISTORY-RETENTION-003.4's per-table independence test: +// office_routine_runs' preview completes successfully, but runs' own preview +// fails in the same sweep (its table is temporarily unreachable). The next +// sweep must not preview office_routine_runs a second time — its preview +// already completed and a sibling's failure must not reopen it — and must +// preview runs again, since its own preview never recorded completion. +func TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + saveZeroFloorSettings(t, sweeper) + + seedRoutine(t, conn, "r-1") + old := daysAgo(60) + seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old) + seedRun(t, conn, newID(), "agent-1", "finished", &old, old) + + // Hide runs after office_routine_runs' own preview work has already + // completed for this sweep, so runs' preview fails on a genuine SQL + // error rather than a simulated one. + testBetweenTablesSweep = func(queryer) { + conn.MustExec(`ALTER TABLE runs RENAME TO runs_hidden`) + } + t.Cleanup(func() { testBetweenTablesSweep = nil }) + + sweeper.RunSweep(ctx) + + first, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if !first.OfficeRoutineRuns.Previewed || first.OfficeRoutineRuns.Err != "" { + t.Fatalf("office_routine_runs = %+v, want a clean, completed preview", first.OfficeRoutineRuns) + } + if first.Runs.Err == "" { + t.Fatal("runs.Err is empty, want the preview failure recorded") + } + if first.Runs.Previewed { + t.Fatal("runs.Previewed = true, want false: the preview did not complete") + } + + testBetweenTablesSweep = nil + conn.MustExec(`ALTER TABLE runs_hidden RENAME TO runs`) + + sweeper.RunSweep(ctx) + + second, ok := sweeper.LastSweepSnapshot() + if !ok { + t.Fatal("LastSweepSnapshot: ok = false, want true") + } + if second.OfficeRoutineRuns.Previewed { + t.Fatal("office_routine_runs.Previewed = true on the second sweep, want false: its preview already completed and must not run a second time because a sibling table failed") + } + if second.OfficeRoutineRuns.Deleted != 1 { + t.Fatalf("office_routine_runs.Deleted = %d, want 1 (it should now be deleting, having already completed its preview)", second.OfficeRoutineRuns.Deleted) + } + if !second.Runs.Previewed || second.Runs.Err != "" { + t.Fatalf("runs = %+v, want a fresh, successful preview: its earlier failed preview must not count as completed", second.Runs) + } + if second.Runs.WouldDelete != 1 { + t.Fatalf("runs.WouldDelete = %d, want 1", second.Runs.WouldDelete) + } +} + +func TestLastSweepSnapshot_FalseBeforeFirstSweep(t *testing.T) { + sweeper, _ := newTestSweeper(t) + if _, ok := sweeper.LastSweepSnapshot(); ok { + t.Fatal("ok = true before any sweep has run, want false") + } +} + +func TestRunCensus_PopulatesRetainedCountsForAllThreeTables(t *testing.T) { + sweeper, conn := newTestSweeper(t) + ctx := context.Background() + + seedRoutine(t, conn, "r-1") + seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1)) + runID := newID() + seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1)) + seedRunEvent(t, conn, runID, 1) + + sweeper.RunCensus(ctx) + + counts := sweeper.CensusSnapshot() + if counts.OfficeRoutineRuns.State != CensusFresh || counts.OfficeRoutineRuns.RetainedCount != 1 { + t.Fatalf("office_routine_runs census = %+v, want fresh with 1", counts.OfficeRoutineRuns) + } + if counts.Runs.State != CensusFresh || counts.Runs.RetainedCount != 1 { + t.Fatalf("runs census = %+v, want fresh with 1", counts.Runs) + } + if counts.RunEvents.State != CensusFresh || counts.RunEvents.RetainedCount != 1 { + t.Fatalf("run_events census = %+v, want fresh with 1", counts.RunEvents) + } +} + +// saveZeroFloorSettings drops both tables' floor to 0 so a test's single +// seeded row is eligible: the default floor of 50 protects the newest 50 +// rows per owner, which a one-row fixture never exceeds. +func saveZeroFloorSettings(t *testing.T, sweeper *Sweeper) { + t.Helper() + settings := DefaultSettings() + settings.RoutineRuns.FloorPerOwner = 0 + settings.Runs.FloorPerOwner = 0 + if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil { + t.Fatalf("SaveSettings: %v", err) + } +} diff --git a/apps/backend/internal/office/retention/types.go b/apps/backend/internal/office/retention/types.go new file mode 100644 index 00000000000..63e7451353b --- /dev/null +++ b/apps/backend/internal/office/retention/types.go @@ -0,0 +1,146 @@ +// Package retention bounds Office run history: office_routine_runs, runs, +// and the run satellite tables (run_events, office_run_route_attempts, +// office_run_skills). See docs/specs/office/requirements/run-history-retention.md +// and its operations counterpart for the frozen contract this package +// implements. +package retention + +import ( + "errors" + "fmt" +) + +// ErrValidation is returned by NormalizeSettings when a field is outside its +// permitted range. The error message names the field. +var ErrValidation = errors.New("retention settings validation") + +// ErrInvalidPersistedSettings wraps a stored settings document that could +// not be read or parsed. Callers fall back to DefaultSettings and surface a +// health issue rather than failing. +var ErrInvalidPersistedSettings = errors.New("invalid persisted retention settings") + +// TableName identifies one of the tables retention reasons about. +type TableName string + +const ( + TableOfficeRoutineRuns TableName = "office_routine_runs" + TableRuns TableName = "runs" + TableRunEvents TableName = "run_events" +) + +// SweptTables are the two tables retention selects rows from by policy, in +// the fixed sweep order. +var SweptTables = []TableName{TableOfficeRoutineRuns, TableRuns} + +// ReportedTables are the swept tables plus the three run satellites that a +// sweep can delete rows from. +var ReportedTables = []TableName{ + TableOfficeRoutineRuns, TableRuns, + "run_events", "office_run_route_attempts", "office_run_skills", +} + +// ThresholdedTables carry a warning threshold on retained row count. +var ThresholdedTables = []TableName{TableOfficeRoutineRuns, TableRuns, TableRunEvents} + +// TableSettings is the per-table retention policy for a swept table. +type TableSettings struct { + WindowDays int `json:"window_days"` + FloorPerOwner int `json:"floor_per_owner"` + WarnRows int `json:"warn_rows"` +} + +// RunEventsSettings is the threshold-only policy for run_events, which has +// no window or floor of its own: its lifetime is its run's. +type RunEventsSettings struct { + WarnRows int `json:"warn_rows"` +} + +// Settings is the full retention policy document, persisted under one +// settings-store key as JSON. +type Settings struct { + Enabled bool `json:"enabled"` + SweepIntervalHours int `json:"sweep_interval_hours"` + BatchLimit int `json:"batch_limit"` + RoutineRuns TableSettings `json:"routine_runs"` + Runs TableSettings `json:"runs"` + RunEvents RunEventsSettings `json:"run_events"` +} + +// DefaultSettings returns the documented AC-OFFICE-RUN-HISTORY-RETENTION-004.2 +// defaults. +func DefaultSettings() Settings { + return Settings{ + Enabled: true, + SweepIntervalHours: 6, + BatchLimit: 5000, + RoutineRuns: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}, + Runs: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}, + RunEvents: RunEventsSettings{WarnRows: 250000}, + } +} + +// Permitted ranges, AC-OFFICE-RUN-HISTORY-RETENTION-004.3. +const ( + minWindowDays = 1 + maxWindowDays = 3650 + + minSweepIntervalHours = 1 + maxSweepIntervalHours = 168 + + minFloorPerOwner = 0 + maxFloorPerOwner = 10000 + + minBatchLimit = 100 + maxBatchLimit = 100000 + + minWarnRows = 0 +) + +// NormalizeSettings validates every field against its permitted range and +// returns a field-named error on the first violation, changing nothing. +// Ranges are inclusive on both ends. +func NormalizeSettings(in Settings) (Settings, error) { + if err := validateRange("sweep_interval_hours", in.SweepIntervalHours, minSweepIntervalHours, maxSweepIntervalHours); err != nil { + return Settings{}, err + } + if err := validateRange("batch_limit", in.BatchLimit, minBatchLimit, maxBatchLimit); err != nil { + return Settings{}, err + } + if err := validateTableSettings("routine_runs", in.RoutineRuns); err != nil { + return Settings{}, err + } + if err := validateTableSettings("runs", in.Runs); err != nil { + return Settings{}, err + } + if err := validateMin("run_events.warn_rows", in.RunEvents.WarnRows, minWarnRows); err != nil { + return Settings{}, err + } + return in, nil +} + +func validateTableSettings(prefix string, s TableSettings) error { + if err := validateRange(prefix+".window_days", s.WindowDays, minWindowDays, maxWindowDays); err != nil { + return err + } + if err := validateRange(prefix+".floor_per_owner", s.FloorPerOwner, minFloorPerOwner, maxFloorPerOwner); err != nil { + return err + } + if err := validateMin(prefix+".warn_rows", s.WarnRows, minWarnRows); err != nil { + return err + } + return nil +} + +func validateRange(field string, value, minValue, maxValue int) error { + if value < minValue || value > maxValue { + return fmt.Errorf("%w: %s must be between %d and %d", ErrValidation, field, minValue, maxValue) + } + return nil +} + +func validateMin(field string, value, minValue int) error { + if value < minValue { + return fmt.Errorf("%w: %s must be %d or greater", ErrValidation, field, minValue) + } + return nil +} diff --git a/apps/backend/internal/office/retention/types_test.go b/apps/backend/internal/office/retention/types_test.go new file mode 100644 index 00000000000..00b61d94d14 --- /dev/null +++ b/apps/backend/internal/office/retention/types_test.go @@ -0,0 +1,94 @@ +package retention + +import "testing" + +func TestDefaultSettings_MatchesDocumentedDefaults(t *testing.T) { + got := DefaultSettings() + + if !got.Enabled { + t.Fatalf("Enabled = false, want true") + } + if got.SweepIntervalHours != 6 { + t.Fatalf("SweepIntervalHours = %d, want 6", got.SweepIntervalHours) + } + if got.BatchLimit != 5000 { + t.Fatalf("BatchLimit = %d, want 5000", got.BatchLimit) + } + wantRoutineRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000} + if got.RoutineRuns != wantRoutineRuns { + t.Fatalf("RoutineRuns = %+v, want %+v", got.RoutineRuns, wantRoutineRuns) + } + wantRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000} + if got.Runs != wantRuns { + t.Fatalf("Runs = %+v, want %+v", got.Runs, wantRuns) + } + if got.RunEvents.WarnRows != 250000 { + t.Fatalf("RunEvents.WarnRows = %d, want 250000", got.RunEvents.WarnRows) + } +} + +func TestNormalizeSettings_AcceptsDefaults(t *testing.T) { + normalized, err := NormalizeSettings(DefaultSettings()) + if err != nil { + t.Fatalf("NormalizeSettings(defaults): %v", err) + } + if normalized != DefaultSettings() { + t.Fatalf("NormalizeSettings(defaults) = %+v, want unchanged defaults", normalized) + } +} + +func TestNormalizeSettings_RejectsOutOfRangeFields(t *testing.T) { + cases := []struct { + name string + mutate func(*Settings) + wantErr string + }{ + {"window too low", func(s *Settings) { s.RoutineRuns.WindowDays = 0 }, "routine_runs.window_days"}, + {"window too high", func(s *Settings) { s.Runs.WindowDays = 3651 }, "runs.window_days"}, + {"interval too low", func(s *Settings) { s.SweepIntervalHours = 0 }, "sweep_interval_hours"}, + {"interval too high", func(s *Settings) { s.SweepIntervalHours = 169 }, "sweep_interval_hours"}, + {"floor negative", func(s *Settings) { s.RoutineRuns.FloorPerOwner = -1 }, "routine_runs.floor_per_owner"}, + {"floor too high", func(s *Settings) { s.Runs.FloorPerOwner = 10001 }, "runs.floor_per_owner"}, + {"batch too low", func(s *Settings) { s.BatchLimit = 99 }, "batch_limit"}, + {"batch too high", func(s *Settings) { s.BatchLimit = 100001 }, "batch_limit"}, + {"warn negative", func(s *Settings) { s.RoutineRuns.WarnRows = -1 }, "routine_runs.warn_rows"}, + {"run_events warn negative", func(s *Settings) { s.RunEvents.WarnRows = -1 }, "run_events.warn_rows"}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + s := DefaultSettings() + tc.mutate(&s) + _, err := NormalizeSettings(s) + if err == nil { + t.Fatalf("NormalizeSettings(%+v): want error, got nil", s) + } + if got := err.Error(); !containsField(got, tc.wantErr) { + t.Fatalf("NormalizeSettings error = %q, want it to name field %q", got, tc.wantErr) + } + }) + } +} + +func TestNormalizeSettings_ZeroWarnRowsDisablesThreshold(t *testing.T) { + s := DefaultSettings() + s.RoutineRuns.WarnRows = 0 + s.Runs.WarnRows = 0 + s.RunEvents.WarnRows = 0 + if _, err := NormalizeSettings(s); err != nil { + t.Fatalf("NormalizeSettings with zero warn thresholds: %v", err) + } +} + +func containsField(msg, field string) bool { + return len(msg) >= len(field) && (indexOf(msg, field) >= 0) +} + +func indexOf(haystack, needle string) int { + for i := 0; i+len(needle) <= len(haystack); i++ { + if haystack[i:i+len(needle)] == needle { + return i + } + } + return -1 +} diff --git a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go index a77e43d3c18..7952730f079 100644 --- a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go +++ b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go @@ -47,9 +47,17 @@ func TestHandleAgentCompleted_FinishRunFailureSkipsCompletionSideEffects(t *test // Force FinishRun's UPDATE to fail with a genuine error, matching // TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails' - // fault injection: a targeted fault (drop the column FinishRun sets) - // rather than a global read-only pragma. - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + // targeted fault rather than a global read-only pragma. + // Block only the terminal timestamp update. The retention expression index + // references finished_at, so dropping the column would fail before the + // handler runs and would no longer exercise its guarded error path. + svc.ExecSQL(t, ` + CREATE TRIGGER block_completion_finish_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "task_id": taskID, @@ -114,7 +122,13 @@ func TestHandleTasklessAgentCompleted_FinishRunFailureSkipsCompletionSideEffects ) `, agent.ID) - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + svc.ExecSQL(t, ` + CREATE TRIGGER block_taskless_completion_finish_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "agent_id": agent.ID, diff --git a/apps/backend/internal/office/service/scheduler_checkout_error_test.go b/apps/backend/internal/office/service/scheduler_checkout_error_test.go index 1cd259ec9ee..05f250781ac 100644 --- a/apps/backend/internal/office/service/scheduler_checkout_error_test.go +++ b/apps/backend/internal/office/service/scheduler_checkout_error_test.go @@ -120,8 +120,17 @@ func TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails(t *testing. // the tasks table, and every other runs column, writable — a targeted // fault instead of a global read-only pragma, so this test actually // distinguishes "release before finish" from "finish before release" - // rather than failing both writes identically. - svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at") + // rather than failing both writes identically. A trigger rather than + // DROP COLUMN: idx_runs_retention is an expression index over + // COALESCE(finished_at, ...), and SQLite refuses to drop a column an + // index still references. + svc.ExecSQL(t, ` + CREATE TRIGGER block_finish_order_test + BEFORE UPDATE OF finished_at ON runs + WHEN NEW.finished_at IS NOT NULL + BEGIN + SELECT RAISE(FAIL, 'finished_at update blocked for test'); + END`) event := bus.NewEvent(events.AgentCompleted, "test", map[string]string{ "task_id": "task-finish-order-1", diff --git a/apps/web/components/settings/system/data-logs-settings.tsx b/apps/web/components/settings/system/data-logs-settings.tsx index 93ff76a4c46..5ef25f4601a 100644 --- a/apps/web/components/settings/system/data-logs-settings.tsx +++ b/apps/web/components/settings/system/data-logs-settings.tsx @@ -7,6 +7,7 @@ import { SettingsTarget } from "@/components/settings/settings-target"; import { BackupsTable } from "@/components/settings/system/backups-table"; import { DatabaseStatsCard } from "@/components/settings/system/database-stats-card"; import { LogViewer } from "@/components/settings/system/log-viewer"; +import { RetentionSettingsCard } from "@/components/settings/system/retention-settings-card"; import { BACKUP_SQL_COMMAND } from "@/components/settings/system/system-route-shell"; import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system"; @@ -35,6 +36,14 @@ export function DataLogsSettings() { + + + + + ({ + fetchRetentionStatus: (...args: unknown[]) => fetchRetentionStatusMock(...args), + saveRetentionSettings: (...args: unknown[]) => saveRetentionSettingsMock(...args), +})); + +vi.mock("@/components/settings/settings-save-provider", () => ({ + useSettingsSaveContributor: (contributor: SettingsSaveContributor) => { + saveContributor = contributor; + }, +})); + +import { RetentionSettingsCard } from "./retention-settings-card"; + +function defaultSettings(overrides: Partial = {}): RetentionSettings { + return { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + ...overrides, + }; +} + +function statusOf(overrides: Partial = {}): RetentionStatus { + return { + settings: defaultSettings(), + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + ...overrides, + }; +} + +function renderCard() { + return render( + + + , + ); +} + +beforeEach(() => { + fetchRetentionStatusMock.mockReset(); + saveRetentionSettingsMock.mockReset(); + fetchRetentionStatusMock.mockResolvedValue(statusOf()); + currentRole = "admin"; + saveContributor = null; +}); + +afterEach(() => { + cleanup(); + vi.clearAllMocks(); +}); + +describe("RetentionSettingsCard", () => { + it("loads settings and renders the swept-table fields from the fetched status", async () => { + renderCard(); + + const windowDays = await screen.findByTestId("retention-routine-runs-window-days"); + expect(windowDays).toHaveProperty("value", "30"); + expect(screen.getByTestId("retention-runs-warn-rows")).toHaveProperty("value", "25000"); + expect(screen.getByTestId("retention-run-events-warn-rows")).toHaveProperty("value", "250000"); + expect(screen.getByTestId("retention-never-swept")).toBeTruthy(); + }); + + it("keeps members read-only while preserving the loaded values", async () => { + currentRole = "member"; + renderCard(); + + const windowDays = await screen.findByTestId("retention-routine-runs-window-days"); + expect(windowDays).toHaveProperty("disabled", true); + expect(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)).toHaveProperty("disabled", true); + expect(screen.getByText("Only an admin can change retention settings.")).toBeTruthy(); + expect(saveContributor?.isDirty).toBe(false); + }); + + it("reports a failed save without clearing the dirty draft", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockRejectedValueOnce(new Error("offline")); + await act(async () => { + await expect(saveContributor?.save(saveContributor.revision)).rejects.toThrow("offline"); + }); + + expect(saveContributor?.isDirty).toBe(true); + await screen.findByTestId("retention-save-error"); + }); + + it("renders the last sweep outcome, backlog flag, and retained counts", async () => { + fetchRetentionStatusMock.mockResolvedValue( + statusOf({ + last_sweep: { + started_at: "2026-09-01T00:00:00Z", + finished_at: "2026-09-01T00:00:05Z", + office_routine_runs: { + deleted: 12, + backlog: true, + error: "", + previewed: false, + would_delete: 0, + }, + runs: { deleted: 3, backlog: false, error: "", previewed: true, would_delete: 40 }, + run_events: { deleted: 100, backlog: false, error: "" }, + route_attempts: { deleted: 0, backlog: false, error: "" }, + run_skills: { deleted: 0, backlog: false, error: "" }, + }, + skip_count: 2, + last_skip_at: "2026-09-01T00:10:00Z", + retained_counts: { + office_routine_runs: { + state: "fresh", + retained_count: 1200, + as_of: "2026-09-01T00:00:00Z", + top_routine_id: "routine-1", + top_routine_share: 0.42, + }, + runs: { state: "stale", retained_count: 800, as_of: "2026-08-31T00:00:00Z" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + renderCard(); + + await screen.findByTestId("retention-last-sweep"); + expect(screen.getByTestId("retention-backlog-office_routine_runs")).toBeTruthy(); + expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain( + "1200", + ); + expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain( + "42%", + ); + expect(screen.getByText(/Stale: last measurement failed/)).toBeTruthy(); + expect(screen.getByTestId("retention-skip-count").textContent).toContain("2"); + }); +}); + +describe("RetentionSettingsCard save/reload consistency", () => { + it("stages an admin edit until the shared save contributor runs, then reloads", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + + const toggle = screen.getByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(toggle); + expect(saveRetentionSettingsMock).not.toHaveBeenCalled(); + expect(saveContributor?.isDirty).toBe(true); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); + fetchRetentionStatusMock.mockResolvedValueOnce( + statusOf({ settings: defaultSettings({ enabled: false }) }), + ); + + await act(async () => saveContributor?.save(saveContributor.revision)); + + expect(saveRetentionSettingsMock).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + ); + await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); + }); + + it("clears the dirty draft from the save response even when the post-save reload fails", async () => { + renderCard(); + await screen.findByTestId(ENABLED_TOGGLE_TEST_ID); + fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)); + if (!saveContributor) throw new Error("expected save contributor"); + + saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false })); + fetchRetentionStatusMock.mockRejectedValueOnce(new Error("offline")); + + await act(async () => saveContributor?.save(saveContributor.revision)); + + expect(saveRetentionSettingsMock).toHaveBeenCalledWith( + expect.objectContaining({ enabled: false }), + ); + await waitFor(() => expect(saveContributor?.isDirty).toBe(false)); + }); +}); + +describe("RetentionSettingsCard unknown-status reporting", () => { + it("renders each unrecognized status with its row count", async () => { + fetchRetentionStatusMock.mockResolvedValue( + statusOf({ + retained_counts: { + office_routine_runs: { + state: "fresh", + retained_count: 5, + as_of: "2026-09-01T00:00:00Z", + unknown_statuses: [{ status: "quarantined", count: 2 }], + }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + renderCard(); + + const row = await screen.findByTestId("retention-retained-office_routine_runs"); + expect(row.textContent).toContain("quarantined (2)"); + }); +}); diff --git a/apps/web/components/settings/system/retention-settings-card.tsx b/apps/web/components/settings/system/retention-settings-card.tsx new file mode 100644 index 00000000000..c08b97fb35f --- /dev/null +++ b/apps/web/components/settings/system/retention-settings-card.tsx @@ -0,0 +1,557 @@ +"use client"; + +import { useEffect, useRef, useState, type ReactNode } from "react"; +import { useTranslation } from "react-i18next"; +import { Alert, AlertDescription } from "@kandev/ui/alert"; +import { CardContent } from "@kandev/ui/card"; +import { Input } from "@kandev/ui/input"; +import { Spinner } from "@kandev/ui/spinner"; +import { Switch } from "@kandev/ui/switch"; +import { IconAlertCircle } from "@tabler/icons-react"; +import { SettingsCard } from "@/components/settings/settings-card"; +import { SettingsCardHeader } from "@/components/settings/settings-card-header"; +import { settingsControlClassName } from "@/components/settings/settings-control"; +import { + SettingsFieldDescription, + SettingsFieldLabel, +} from "@/components/settings/settings-typography"; +import { useSettingsSaveContributor } from "@/components/settings/settings-save-provider"; +import { useIsAdmin } from "@/hooks/domains/auth/use-is-admin"; +import { useRetentionSettings } from "@/hooks/domains/system/use-retention-settings"; +import { formatDateTime } from "@/lib/i18n/formats"; +import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system"; +import type { + RetentionSettings, + RetentionStatus, + RetentionTableCensus, + RetentionTableSweepResult, + RetentionSweptTableResult, + RetentionUnknownStatusCount, +} from "@/lib/types/system"; + +function serialize(settings: RetentionSettings | null): string { + return settings ? JSON.stringify(settings) : "loading"; +} + +function formatUnknownStatuses(unknown: RetentionUnknownStatusCount[]): string { + return unknown.map((u) => `${u.status} (${u.count})`).join(", "); +} + +function NumberField({ + label, + help, + value, + min, + max, + disabled, + onChange, + testId, +}: { + label: string; + help: string; + value: number; + min: number; + max?: number; + disabled?: boolean; + onChange: (value: number) => void; + testId: string; +}) { + return ( +
+ {label} + onChange(Number(event.target.value))} + className={settingsControlClassName("h-11")} + data-testid={testId} + /> + {help} +
+ ); +} + +function useRetentionDraft(remote: ReturnType, isAdmin: boolean) { + const { t } = useTranslation(); + const [draft, setDraft] = useState(null); + const previousSaved = useRef(null); + const saved = remote.status?.settings ?? null; + + useEffect(() => { + if (!saved) return; + setDraft((current) => { + const previous = previousSaved.current; + if (!current || !previous || serialize(current) === serialize(previous)) return saved; + return current; + }); + previousSaved.current = saved; + }, [saved]); + + const isDirty = Boolean(draft && saved && serialize(draft) !== serialize(saved)); + const canEdit = isAdmin && !remote.isLoading && Boolean(saved); + const invalidReason = !isAdmin ? t("system:retentionAdminOnly") : undefined; + + useSettingsSaveContributor({ + id: "system:retention", + order: 25, + revision: serialize(draft), + isDirty, + canSave: canEdit, + invalidReason, + save: async () => { + if (!draft) return; + await remote.save(draft); + }, + discard: () => { + if (saved) setDraft(saved); + }, + }); + + return { draft, setDraft, saved, canEdit }; +} + +function RetentionEnabledRow({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( +
+
+ + {t("system:retentionEnabledLabel")} + + + {t("system:retentionEnabledDescription")} + +
+ onChange({ ...settings, enabled })} + data-testid="retention-enabled" + className="shrink-0 cursor-pointer" + /> +
+ ); +} + +function RetentionScheduleFields({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( +
+ onChange({ ...settings, sweep_interval_hours })} + testId="retention-sweep-interval" + /> + onChange({ ...settings, batch_limit })} + testId="retention-batch-limit" + /> +
+ ); +} + +function TableSection({ + title, + description, + children, +}: { + title: string; + description: string; + children: ReactNode; +}) { + return ( +
+
+

{title}

+

{description}

+
+
{children}
+
+ ); +} + +type WindowedTableKey = "routine_runs" | "runs"; + +function WindowedTableSection({ + tableKey, + title, + description, + settings, + disabled, + onChange, +}: { + tableKey: WindowedTableKey; + title: string; + description: string; + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + const table = settings[tableKey]; + const testPrefix = tableKey === "routine_runs" ? "retention-routine-runs" : "retention-runs"; + return ( + + onChange({ ...settings, [tableKey]: { ...table, window_days } })} + testId={`${testPrefix}-window-days`} + /> + + onChange({ ...settings, [tableKey]: { ...table, floor_per_owner } }) + } + testId={`${testPrefix}-floor-per-owner`} + /> + onChange({ ...settings, [tableKey]: { ...table, warn_rows } })} + testId={`${testPrefix}-warn-rows`} + /> + + ); +} + +function RunEventsSection({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( + + onChange({ ...settings, run_events: { warn_rows } })} + testId="retention-run-events-warn-rows" + /> + + ); +} + +function RoutineRunsAndRunsSections({ + settings, + disabled, + onChange, +}: { + settings: RetentionSettings; + disabled: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + return ( + <> + + + + + ); +} + +function RetentionPolicyCard({ + draft, + canEdit, + onChange, +}: { + draft: RetentionSettings; + canEdit: boolean; + onChange: (settings: RetentionSettings) => void; +}) { + const { t } = useTranslation(); + const disabled = !canEdit; + return ( + + + + + + + {!canEdit && ( +

{t("system:retentionAdminOnly")}

+ )} +
+
+ ); +} + +function SweptTableRow({ label, result }: { label: string; result: RetentionSweptTableResult }) { + const { t } = useTranslation(); + return ( +
+ {label} + + {t("system:retentionDeletedLabel")}: {result.deleted} + + {result.previewed && ( + + {t("system:retentionWouldDeleteLabel")}: {result.would_delete} + + )} + {result.backlog && ( + + {t("system:retentionBacklogLabel")} + + )} + {result.error && ( + + {t("system:retentionTableErrorLabel")}: {result.error} + + )} +
+ ); +} + +function SatelliteTableRow({ + label, + result, +}: { + label: string; + result: RetentionTableSweepResult; +}) { + const { t } = useTranslation(); + return ( +
+ {label} + + {t("system:retentionDeletedLabel")}: {result.deleted} + + {result.error && ( + + {t("system:retentionTableErrorLabel")}: {result.error} + + )} +
+ ); +} + +function LastSweepSection({ status }: { status: RetentionStatus }) { + const { t } = useTranslation(); + const lastSweep = status.last_sweep; + if (!lastSweep) { + return ( +

+ {t("system:retentionNeverSweptMessage")} +

+ ); + } + return ( +
+

+ {t("system:retentionSweepStartedAtLabel")}: {formatDateTime(lastSweep.started_at)} + {" · "} + {t("system:retentionSweepFinishedAtLabel")}: {formatDateTime(lastSweep.finished_at)} +

+ + + + + +
+ ); +} + +function RetainedCountRow({ label, census }: { label: string; census: RetentionTableCensus }) { + const { t } = useTranslation(); + if (census.state === "not_computed") { + return ( +
+ {label} + {t("system:retentionCensusNotComputed")} +
+ ); + } + return ( +
+ {label} + {census.retained_count} + + {t("system:retentionCensusAsOfLabel")}: {formatDateTime(census.as_of)} + + {census.state === "stale" && ( + {t("system:retentionCensusStale")} + )} + {census.unknown_statuses && census.unknown_statuses.length > 0 && ( + + {t("system:retentionUnknownStatusesLabel")}:{" "} + {formatUnknownStatuses(census.unknown_statuses)} + + )} + {census.top_routine_id && ( + + {t("system:retentionTopRoutineShareLabel")}:{" "} + {Math.round((census.top_routine_share ?? 0) * 100)}% ({census.top_routine_id}) + + )} +
+ ); +} + +function RetentionStatusCard({ status }: { status: RetentionStatus | null }) { + const { t } = useTranslation(); + if (!status) return null; + return ( + + + + +
+

{t("system:retentionRetainedCountsTitle")}

+ + + +
+ {status.skip_count > 0 && ( +

+ {t("system:retentionSkipCountLabel")}: {status.skip_count} + {status.last_skip_at && ( + <> + {" · "} + {t("system:retentionLastSkipAtLabel")}: {formatDateTime(status.last_skip_at)} + + )} +

+ )} +
+
+ ); +} + +function RetentionSettingsLoading() { + const { t } = useTranslation(); + return ( + + + + {t("settings:loading")} + + + ); +} + +function RetentionSettingsLoadError({ error }: { error: string }) { + const { t } = useTranslation(); + return ( + + + + {t("system:retentionLoadFailed")}: {error} + + + ); +} + +export function RetentionSettingsCard() { + const remote = useRetentionSettings(); + const isAdmin = useIsAdmin(); + const { draft, setDraft, canEdit } = useRetentionDraft(remote, isAdmin); + const { t } = useTranslation(); + + if (remote.isLoading && !remote.status) return ; + if (remote.error && !remote.status) return ; + + return ( +
+ {draft && } + + {remote.saveError && ( + + + + {t("system:retentionSaveFailed")}: {remote.saveError} + + + )} +
+ ); +} diff --git a/apps/web/components/settings/system/system-route-copy.test.ts b/apps/web/components/settings/system/system-route-copy.test.ts index 669405ac22c..1024b5e15df 100644 --- a/apps/web/components/settings/system/system-route-copy.test.ts +++ b/apps/web/components/settings/system/system-route-copy.test.ts @@ -17,6 +17,7 @@ vi.mock("@/components/settings/settings-target", () => ({ vi.mock("./backups-table", () => ({ BackupsTable: () => null })); vi.mock("./database-stats-card", () => ({ DatabaseStatsCard: () => null })); vi.mock("./log-viewer", () => ({ LogViewer: () => null })); +vi.mock("./retention-settings-card", () => ({ RetentionSettingsCard: () => null })); afterEach(() => { cleanup(); databaseState.value = null; @@ -147,6 +148,7 @@ describe("Data & Logs composition", () => { render(createElement(DataLogsSettings)); expect(screen.getByText(t("system:navDatabase"))).toBeTruthy(); + expect(screen.getByText(t("system:navRetention"))).toBeTruthy(); expect(screen.getByText(t("system:navBackups"))).toBeTruthy(); expect(screen.getByText(t("system:navLogs"))).toBeTruthy(); expect(screen.queryByText(t("system:storageTitle"))).toBeNull(); diff --git a/apps/web/e2e/tests/system/retention-settings.spec.ts b/apps/web/e2e/tests/system/retention-settings.spec.ts new file mode 100644 index 00000000000..08a75c77394 --- /dev/null +++ b/apps/web/e2e/tests/system/retention-settings.spec.ts @@ -0,0 +1,97 @@ +import { test, expect } from "../../fixtures/test-base"; + +/** + * Office run-history retention (docs/specs/office/requirements/run-history-retention*.md). + * The sweep itself is time- and seed-dependent and is covered far more cheaply by backend + * tests; the two honest E2E candidates are the settings round-trip through the real API + * and a threshold warning actually reaching the Health card. Both tests restore whatever + * global retention settings they change, since retention settings are process-global and + * this worker's backend is reused by every test file in the shard. + */ +test.describe("System retention settings", () => { + test("persists an admin policy edit through save and across a backend restart", async ({ + testPage, + backend, + }) => { + test.setTimeout(90_000); + await testPage.goto("/settings/system/data-storage"); + const batchLimit = testPage.getByTestId("retention-batch-limit"); + await expect(batchLimit).toBeVisible(); + const original = await batchLimit.inputValue(); + const updated = original === "7777" ? "8888" : "7777"; + + try { + await batchLimit.fill(updated); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + + await testPage.reload(); + await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated); + + await backend.restart(); + await testPage.reload(); + await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated); + } finally { + await testPage.getByTestId("retention-batch-limit").fill(original); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + } + }); + + test("shows a threshold warning on the Health card once retained runs exceed the configured limit", async ({ + testPage, + backend, + apiClient, + seedData, + }) => { + test.setTimeout(90_000); + await testPage.goto("/settings/system/data-storage"); + const warnField = testPage.getByTestId("retention-runs-warn-rows"); + await expect(warnField).toBeVisible(); + const originalWarnRows = await warnField.inputValue(); + + const initialStatus = await testPage.evaluate(async () => { + const response = await fetch("/api/v1/system/retention"); + return response.json(); + }); + const baselineRetained = + initialStatus.retained_counts.runs.state === "fresh" + ? initialStatus.retained_counts.runs.retained_count + : 0; + // warn_rows=0 disables the threshold check entirely (AC-OFFICE-RUN-HISTORY-RETENTION-004.3), + // and seeding 3 new terminal runs guarantees the post-seed count clears baseline+1 even when + // baseline is 0. + const seededRunCount = 3; + const newWarnRows = baselineRetained + 1; + const expectedRetained = baselineRetained + seededRunCount; + for (let i = 0; i < seededRunCount; i++) { + await apiClient.seedRun({ agentProfileId: seedData.agentProfileId, status: "finished" }); + } + + try { + await warnField.fill(String(newWarnRows)); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + + // The census only re-evaluates on its interval timer or at scheduler Start; a + // restart forces an immediate re-evaluation against the settings just saved and + // the runs just seeded, deterministically, without waiting out the real interval. + await backend.restart(); + + await testPage.goto("/settings/system/status"); + const issue = testPage.getByTestId("system-health-issue-office_retention_threshold:runs"); + await expect(issue).toBeVisible({ timeout: 15_000 }); + await expect(issue).toContainText("over its threshold"); + + await testPage.goto("/settings/system/data-storage"); + const retainedRuns = testPage.getByTestId("retention-retained-runs"); + await expect(retainedRuns).toBeVisible(); + await expect(retainedRuns).toContainText(String(expectedRetained)); + } finally { + await testPage.goto("/settings/system/data-storage"); + await testPage.getByTestId("retention-runs-warn-rows").fill(originalWarnRows); + await testPage.getByRole("button", { name: "Save changes" }).click(); + await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved"); + } + }); +}); diff --git a/apps/web/hooks/domains/system/use-retention-settings.ts b/apps/web/hooks/domains/system/use-retention-settings.ts new file mode 100644 index 00000000000..d93f2d6deb6 --- /dev/null +++ b/apps/web/hooks/domains/system/use-retention-settings.ts @@ -0,0 +1,67 @@ +"use client"; + +import { useCallback, useEffect, useState } from "react"; +import { useAppStore, useAppStoreApi } from "@/components/state-provider"; +import { fetchRetentionStatus, saveRetentionSettings } from "@/lib/api/domains/system-api"; +import type { RetentionSettings } from "@/lib/types/system"; + +export function useRetentionSettings() { + const status = useAppStore((s) => s.system.retention); + const setStatus = useAppStore((s) => s.setSystemRetention); + const storeApi = useAppStoreApi(); + const [isLoading, setIsLoading] = useState(false); + const [error, setError] = useState(null); + const [saveError, setSaveError] = useState(null); + + const reload = useCallback(async () => { + setIsLoading(true); + setError(null); + try { + setStatus(await fetchRetentionStatus({ cache: "no-store" })); + } catch (e) { + setError(e instanceof Error ? e.message : String(e)); + } finally { + setIsLoading(false); + } + }, [setStatus]); + + useEffect(() => { + if (status) return; + void reload(); + }, [status, reload]); + + const save = useCallback( + async (settings: RetentionSettings) => { + setSaveError(null); + try { + const saved = await saveRetentionSettings(settings); + // Apply the PUT's own normalized response synchronously, rather + // than relying solely on the reload() below: if that GET fails, + // its error is recorded but never surfaces once status is already + // loaded (see the isLoading/error-gated branches in + // RetentionSettingsCard), which would otherwise leave the store + // holding pre-save settings while the save coordinator believes + // the save already succeeded. + storeApi.setState((state) => + state.system.retention + ? { + system: { + ...state.system, + retention: { ...state.system.retention, settings: saved }, + }, + } + : state, + ); + void reload(); + return saved; + } catch (e) { + const message = e instanceof Error ? e.message : String(e); + setSaveError(message); + throw e; + } + }, + [reload, storeApi], + ); + + return { status, isLoading, error, saveError, reload, save }; +} diff --git a/apps/web/lib/api/domains/system-api.test.ts b/apps/web/lib/api/domains/system-api.test.ts index c1980b1dd63..fda9156ac2c 100644 --- a/apps/web/lib/api/domains/system-api.test.ts +++ b/apps/web/lib/api/domains/system-api.test.ts @@ -43,6 +43,8 @@ import { restoreStorageQuarantine, runStorageMaintenance, saveStorageSettings, + fetchRetentionStatus, + saveRetentionSettings, } from "./system-api"; const BASE = "http://api.test/api/v1/system"; @@ -528,3 +530,47 @@ describe("storage policy", () => { expect(response.capabilities.docker_available).toBe(true); }); }); + +describe("office run history retention", () => { + const retentionSettings = { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + }; + + it("loads retention status without caching", async () => { + fetchSpy.mockResolvedValueOnce( + jsonResponse({ + settings: retentionSettings, + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" }, + runs: { state: "not_computed", retained_count: 0, as_of: "" }, + run_events: { state: "not_computed", retained_count: 0, as_of: "" }, + }, + }), + ); + + const response = await fetchRetentionStatus(); + + expect(lastCall().url).toBe(`${BASE}/retention`); + expect(lastCall().init?.cache).toBe("no-store"); + expect(response.settings).toEqual(retentionSettings); + expect(response.last_sweep).toBeNull(); + }); + + it("PUTs the full settings document to save", async () => { + fetchSpy.mockResolvedValueOnce(jsonResponse(retentionSettings)); + + const response = await saveRetentionSettings(retentionSettings); + + expect(lastCall().url).toBe(`${BASE}/retention`); + expect(method()).toBe("PUT"); + expect(JSON.parse(String(lastCall().init?.body))).toEqual(retentionSettings); + expect(response).toEqual(retentionSettings); + }); +}); diff --git a/apps/web/lib/api/domains/system-api.ts b/apps/web/lib/api/domains/system-api.ts index 8c27004e7b8..4c5395cfd27 100644 --- a/apps/web/lib/api/domains/system-api.ts +++ b/apps/web/lib/api/domains/system-api.ts @@ -24,6 +24,8 @@ import type { StorageQuarantinePurgeScope, StorageSettingsResponse, UpdatesChannel, + RetentionSettings, + RetentionStatus, } from "@/lib/types/system"; const SYSTEM_BASE = "/api/v1/system"; @@ -416,3 +418,26 @@ export function purgeStorageQuarantine( }, }); } + +// --- Office run history retention ---------------------------------------- + +export function fetchRetentionStatus(options?: ApiRequestOptions): Promise { + return fetchJson(`${SYSTEM_BASE}/retention`, { + ...options, + cache: "no-store", + }); +} + +export function saveRetentionSettings( + settings: RetentionSettings, + options?: ApiRequestOptions, +): Promise { + return fetchJson(`${SYSTEM_BASE}/retention`, { + ...options, + init: { + ...(options?.init ?? {}), + method: "PUT", + body: JSON.stringify(settings), + }, + }); +} diff --git a/apps/web/lib/settings-discovery/catalog/system.ts b/apps/web/lib/settings-discovery/catalog/system.ts index f2f71da4bcc..82405c65736 100644 --- a/apps/web/lib/settings-discovery/catalog/system.ts +++ b/apps/web/lib/settings-discovery/catalog/system.ts @@ -9,6 +9,7 @@ export const SYSTEM_STORAGE_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/storage`; export const SYSTEM_ABOUT_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/about`; export const SYSTEM_SETTINGS_TARGETS = { database: "setting-system-database", + retention: "setting-system-retention", backups: "setting-system-backups", logs: "setting-system-logs", licenses: "setting-system-licenses", @@ -59,6 +60,16 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ targetId: SYSTEM_SETTINGS_TARGETS.database, order: 621, }, + { + id: "system-retention", + kind: "section", + labelKey: "system:navRetention", + parentId: SYSTEM_DATA_STORAGE_DISCOVERY_ID, + groupId: "system", + href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, + targetId: SYSTEM_SETTINGS_TARGETS.retention, + order: 622, + }, { id: "system-backups", kind: "section", @@ -67,7 +78,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ groupId: "system", href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, targetId: SYSTEM_SETTINGS_TARGETS.backups, - order: 622, + order: 623, }, { id: "system-logs", @@ -77,7 +88,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [ groupId: "system", href: SYSTEM_DATA_STORAGE_SETTINGS_HREF, targetId: SYSTEM_SETTINGS_TARGETS.logs, - order: 623, + order: 624, }, { id: "system-storage", diff --git a/apps/web/lib/state/slices/system/system-slice.test.ts b/apps/web/lib/state/slices/system/system-slice.test.ts index ef24efea1e4..00e1103e73e 100644 --- a/apps/web/lib/state/slices/system/system-slice.test.ts +++ b/apps/web/lib/state/slices/system/system-slice.test.ts @@ -200,6 +200,30 @@ describe("system slice", () => { expect(store.getState().system.database).toEqual(DB_STATS); }); + it("setSystemRetention stores the status", () => { + const store = makeStore(); + expect(store.getState().system.retention).toBeNull(); + const status = { + settings: { + enabled: true, + sweep_interval_hours: 6, + batch_limit: 5000, + routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 }, + run_events: { warn_rows: 250000 }, + }, + last_sweep: null, + skip_count: 0, + retained_counts: { + office_routine_runs: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + runs: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + run_events: { state: "not_computed" as const, retained_count: 0, as_of: "" }, + }, + }; + store.getState().setSystemRetention(status); + expect(store.getState().system.retention).toEqual(status); + }); + it("setSystemBackups marks the list as loaded", () => { const store = makeStore(); store.getState().setSystemBackups([SNAPSHOT]); diff --git a/apps/web/lib/state/slices/system/system-slice.ts b/apps/web/lib/state/slices/system/system-slice.ts index 22ed5baeaa8..fc26162b758 100644 --- a/apps/web/lib/state/slices/system/system-slice.ts +++ b/apps/web/lib/state/slices/system/system-slice.ts @@ -6,6 +6,7 @@ export const defaultSystemState: SystemSliceState = { info: null, diskUsage: null, database: null, + retention: null, backups: { items: [], loaded: false }, updates: null, jobs: {}, @@ -44,6 +45,10 @@ export const createSystemSlice: StateCreator< set((draft) => { draft.system.database = stats; }), + setSystemRetention: (status) => + set((draft) => { + draft.system.retention = status; + }), setSystemBackups: (items) => set((draft) => { draft.system.backups = { items, loaded: true }; diff --git a/apps/web/lib/state/slices/system/types.ts b/apps/web/lib/state/slices/system/types.ts index 7135fdaf363..ecb21316e54 100644 --- a/apps/web/lib/state/slices/system/types.ts +++ b/apps/web/lib/state/slices/system/types.ts @@ -11,6 +11,7 @@ import type { StorageOverviewResponse, StoragePolicyResponse, StorageQuarantineEntry, + RetentionStatus, } from "@/lib/types/system"; export type SystemBackupsState = { @@ -25,6 +26,7 @@ export type SystemSliceState = { info: SystemInfo | null; diskUsage: DiskUsageResponse | null; database: DatabaseStats | null; + retention: RetentionStatus | null; backups: SystemBackupsState; updates: UpdatesResponse | null; jobs: SystemJobsMap; @@ -44,6 +46,7 @@ export type SystemSliceActions = { setSystemInfo: (info: SystemInfo) => void; setSystemDiskUsage: (usage: DiskUsageResponse) => void; setSystemDatabase: (stats: DatabaseStats) => void; + setSystemRetention: (status: RetentionStatus) => void; setSystemBackups: (items: SnapshotInfo[]) => void; setSystemUpdates: (updates: UpdatesResponse) => void; upsertSystemJob: (job: SystemJob) => void; diff --git a/apps/web/lib/types/system.ts b/apps/web/lib/types/system.ts index 09949372fbb..399bc00dd05 100644 --- a/apps/web/lib/types/system.ts +++ b/apps/web/lib/types/system.ts @@ -560,6 +560,78 @@ export interface StorageAdoptionResponse extends StorageSettingsResponse { capabilities: StorageCapabilities; } +// --- Office run history retention --------------------------------------- + +export interface RetentionTableSettings { + window_days: number; + floor_per_owner: number; + warn_rows: number; +} + +export interface RetentionRunEventsSettings { + warn_rows: number; +} + +export interface RetentionSettings { + enabled: boolean; + sweep_interval_hours: number; + batch_limit: number; + routine_runs: RetentionTableSettings; + runs: RetentionTableSettings; + run_events: RetentionRunEventsSettings; +} + +export interface RetentionTableSweepResult { + deleted: number; + backlog: boolean; + error: string; +} + +export interface RetentionSweptTableResult extends RetentionTableSweepResult { + previewed: boolean; + would_delete: number; +} + +export interface RetentionLastSweep { + started_at: string; + finished_at: string; + office_routine_runs: RetentionSweptTableResult; + runs: RetentionSweptTableResult; + run_events: RetentionTableSweepResult; + route_attempts: RetentionTableSweepResult; + run_skills: RetentionTableSweepResult; +} + +export type RetentionCensusState = "not_computed" | "fresh" | "stale"; + +export interface RetentionUnknownStatusCount { + status: string; + count: number; +} + +export interface RetentionTableCensus { + state: RetentionCensusState; + retained_count: number; + as_of: string; + unknown_statuses?: RetentionUnknownStatusCount[]; + top_routine_id?: string; + top_routine_share?: number; +} + +export interface RetentionRetainedCounts { + office_routine_runs: RetentionTableCensus; + runs: RetentionTableCensus; + run_events: RetentionTableCensus; +} + +export interface RetentionStatus { + settings: RetentionSettings; + last_sweep: RetentionLastSweep | null; + skip_count: number; + last_skip_at?: string; + retained_counts: RetentionRetainedCounts; +} + export interface RestartCapability { supported: boolean; mode: "manual" | "supervisor" | string; diff --git a/apps/web/src/locales/en/system.json b/apps/web/src/locales/en/system.json index 63bda933697..b103768c767 100644 --- a/apps/web/src/locales/en/system.json +++ b/apps/web/src/locales/en/system.json @@ -207,8 +207,52 @@ "navFeatureToggles": "Feature Toggles", "navLicenses": "Licenses", "navLogs": "Logs", + "navRetention": "Run History Retention", "navUpdates": "Updates", "navUsers": "Users", + "retentionPageDescription": "Bounds how long Office run history (office_routine_runs, runs, and their satellite tables) stays in the database.", + "retentionAdminOnly": "Only an admin can change retention settings.", + "retentionLoadFailed": "Retention settings could not be loaded.", + "retentionSaveFailed": "Retention settings could not be saved.", + "retentionPolicyTitle": "Retention policy", + "retentionPolicyDescription": "A scheduled sweep, separate from the 5-second Office tick, deletes finished run history once it passes its retention window. Rows that represent current work in progress are never deleted, regardless of these settings.", + "retentionEnabledLabel": "Delete eligible run history", + "retentionEnabledDescription": "When off, Kandev never deletes office_routine_runs or runs rows. Retained-row counts below keep updating so you can see backlog build up before turning deletion back on.", + "retentionSweepIntervalLabel": "Sweep interval (hours)", + "retentionSweepIntervalHelp": "How often the sweep runs, from 1 to 168 hours (1 week).", + "retentionBatchLimitLabel": "Rows deleted per sweep", + "retentionBatchLimitHelp": "Maximum rows deleted per table in one sweep, from 100 to 100,000. Lower this if a sweep competes with other database load; a large backlog is cleared over several sweeps instead of one.", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "Routine execution records: one row per completed or failed routine run.", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "Task run records: one row per completed, failed, or cancelled run.", + "retentionRunEventsSectionTitle": "run_events and satellite tables", + "retentionRunEventsSectionDescription": "run_events, office_run_route_attempts, and office_run_skills rows are deleted together with the run that owns them; they have no window or floor of their own.", + "retentionWindowDaysLabel": "Retention window (days)", + "retentionWindowDaysHelp": "Finished rows older than this many days become eligible for deletion, from 1 to 3650 days.", + "retentionFloorPerOwnerLabel": "Minimum kept per owner", + "retentionFloorPerOwnerHelp": "Always keeps at least this many of each owner's most recent finished rows, even past the retention window, from 0 to 10,000. Set to 0 to disable the floor.", + "retentionWarnRowsLabel": "Warn above this many retained rows", + "retentionWarnRowsHelp": "Raises a health warning once retained rows exceed this count. Set to 0 to disable the warning.", + "retentionRunEventsWarnRowsHelp": "Raises a health warning once total retained run_events rows exceed this count. Set to 0 to disable the warning.", + "retentionStatusTitle": "Retention status", + "retentionStatusDescription": "The most recent sweep outcome and the current retained-row counts for each thresholded table.", + "retentionNeverSweptMessage": "No sweep has run yet since the backend last started.", + "retentionSweepStartedAtLabel": "Started", + "retentionSweepFinishedAtLabel": "Finished", + "retentionDeletedLabel": "Deleted", + "retentionWouldDeleteLabel": "Would delete (preview)", + "retentionBacklogLabel": "Backlog: more eligible rows remain past this sweep's batch limit", + "retentionTableErrorLabel": "Error", + "retentionSkipCountLabel": "Sweeps skipped", + "retentionSkipCountHelp": "Counts sweeps that were due but did not run, most often because another Kandev process held the retention lock at the same time.", + "retentionLastSkipAtLabel": "Last skipped", + "retentionRetainedCountsTitle": "Retained rows", + "retentionCensusNotComputed": "Not yet measured", + "retentionCensusStale": "Stale: last measurement failed, showing the last successful count", + "retentionCensusAsOfLabel": "As of", + "retentionUnknownStatusesLabel": "Rows with an unrecognized status (not counted as history or live state)", + "retentionTopRoutineShareLabel": "Largest single routine's share of retained rows", "backendReloadRequiredTitle": "Reload required", "backendReloadRequiredBody": "Kandev restarted. Reload this page to continue. Reloading discards unsaved changes.", "backendReloadRequiredAction": "Reload page", diff --git a/apps/web/src/locales/pseudo/system.json b/apps/web/src/locales/pseudo/system.json index 7711697a8e3..f90f569e914 100644 --- a/apps/web/src/locales/pseudo/system.json +++ b/apps/web/src/locales/pseudo/system.json @@ -207,8 +207,52 @@ "navFeatureToggles": "Ƒēàţũŕē Ţōĝĝĺēś", "navLicenses": "Ĺĩćēńśēś", "navLogs": "Ĺōĝś", + "navRetention": "Ŕũń Ĥĩśţōŕŷ Ŕēţēńţĩōń", "navUpdates": "Ũƥďàţēś", "navUsers": "Ũśēŕś", + "retentionPageDescription": "Ɓōũńďś ĥōŵ ĺōńĝ Ōƒƒĩćē ŕũń ĥĩśţōŕŷ (ōƒƒĩćē_ŕōũţĩńē_ŕũńś, ŕũńś, àńď ţĥēĩŕ śàţēĺĺĩţē ţàƀĺēś) śţàŷś ĩń ţĥē ďàţàƀàśē.", + "retentionAdminOnly": "Ōńĺŷ àń àďḿĩń ćàń ćĥàńĝē ŕēţēńţĩōń śēţţĩńĝś.", + "retentionLoadFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē ĺōàďēď.", + "retentionSaveFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē śàvēď.", + "retentionPolicyTitle": "Ŕēţēńţĩōń ƥōĺĩćŷ", + "retentionPolicyDescription": "À śćĥēďũĺēď śŵēēƥ, śēƥàŕàţē ƒŕōḿ ţĥē 5-śēćōńď Ōƒƒĩćē ţĩćķ, ďēĺēţēś ƒĩńĩśĥēď ŕũń ĥĩśţōŕŷ ōńćē ĩţ ƥàśśēś ĩţś ŕēţēńţĩōń ŵĩńďōŵ. Ŕōŵś ţĥàţ ŕēƥŕēśēńţ ćũŕŕēńţ ŵōŕķ ĩń ƥŕōĝŕēśś àŕē ńēvēŕ ďēĺēţēď, ŕēĝàŕďĺēśś ōƒ ţĥēśē śēţţĩńĝś.", + "retentionEnabledLabel": "Ďēĺēţē ēĺĩĝĩƀĺē ŕũń ĥĩśţōŕŷ", + "retentionEnabledDescription": "Ŵĥēń ōƒƒ, Ķàńďēv ńēvēŕ ďēĺēţēś ōƒƒĩćē_ŕōũţĩńē_ŕũńś ōŕ ŕũńś ŕōŵś. Ŕēţàĩńēď-ŕōŵ ćōũńţś ƀēĺōŵ ķēēƥ ũƥďàţĩńĝ śō ŷōũ ćàń śēē ƀàćķĺōĝ ƀũĩĺď ũƥ ƀēƒōŕē ţũŕńĩńĝ ďēĺēţĩōń ƀàćķ ōń.", + "retentionSweepIntervalLabel": "Śŵēēƥ ĩńţēŕvàĺ (ĥōũŕś)", + "retentionSweepIntervalHelp": "Ĥōŵ ōƒţēń ţĥē śŵēēƥ ŕũńś, ƒŕōḿ 1 ţō 168 ĥōũŕś (1 ŵēēķ).", + "retentionBatchLimitLabel": "Ŕōŵś ďēĺēţēď ƥēŕ śŵēēƥ", + "retentionBatchLimitHelp": "Ḿàxĩḿũḿ ŕōŵś ďēĺēţēď ƥēŕ ţàƀĺē ĩń ōńē śŵēēƥ, ƒŕōḿ 100 ţō 100,000. Ĺōŵēŕ ţĥĩś ĩƒ à śŵēēƥ ćōḿƥēţēś ŵĩţĥ ōţĥēŕ ďàţàƀàśē ĺōàď; à ĺàŕĝē ƀàćķĺōĝ ĩś ćĺēàŕēď ōvēŕ śēvēŕàĺ śŵēēƥś ĩńśţēàď ōƒ ōńē.", + "retentionRoutineRunsSectionTitle": "ōƒƒĩćē_ŕōũţĩńē_ŕũńś", + "retentionRoutineRunsSectionDescription": "Ŕōũţĩńē ēxēćũţĩōń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď ōŕ ƒàĩĺēď ŕōũţĩńē ŕũń.", + "retentionRunsSectionTitle": "ŕũńś", + "retentionRunsSectionDescription": "Ţàśķ ŕũń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď, ƒàĩĺēď, ōŕ ćàńćēĺĺēď ŕũń.", + "retentionRunEventsSectionTitle": "ŕũń_ēvēńţś àńď śàţēĺĺĩţē ţàƀĺēś", + "retentionRunEventsSectionDescription": "ŕũń_ēvēńţś, ōƒƒĩćē_ŕũń_ŕōũţē_àţţēḿƥţś, àńď ōƒƒĩćē_ŕũń_śķĩĺĺś ŕōŵś àŕē ďēĺēţēď ţōĝēţĥēŕ ŵĩţĥ ţĥē ŕũń ţĥàţ ōŵńś ţĥēḿ; ţĥēŷ ĥàvē ńō ŵĩńďōŵ ōŕ ƒĺōōŕ ōƒ ţĥēĩŕ ōŵń.", + "retentionWindowDaysLabel": "Ŕēţēńţĩōń ŵĩńďōŵ (ďàŷś)", + "retentionWindowDaysHelp": "Ƒĩńĩśĥēď ŕōŵś ōĺďēŕ ţĥàń ţĥĩś ḿàńŷ ďàŷś ƀēćōḿē ēĺĩĝĩƀĺē ƒōŕ ďēĺēţĩōń, ƒŕōḿ 1 ţō 3650 ďàŷś.", + "retentionFloorPerOwnerLabel": "Ḿĩńĩḿũḿ ķēƥţ ƥēŕ ōŵńēŕ", + "retentionFloorPerOwnerHelp": "Àĺŵàŷś ķēēƥś àţ ĺēàśţ ţĥĩś ḿàńŷ ōƒ ēàćĥ ōŵńēŕ'ś ḿōśţ ŕēćēńţ ƒĩńĩśĥēď ŕōŵś, ēvēń ƥàśţ ţĥē ŕēţēńţĩōń ŵĩńďōŵ, ƒŕōḿ 0 ţō 10,000. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ƒĺōōŕ.", + "retentionWarnRowsLabel": "Ŵàŕń àƀōvē ţĥĩś ḿàńŷ ŕēţàĩńēď ŕōŵś", + "retentionWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ŕēţàĩńēď ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.", + "retentionRunEventsWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ţōţàĺ ŕēţàĩńēď ŕũń_ēvēńţś ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.", + "retentionStatusTitle": "Ŕēţēńţĩōń śţàţũś", + "retentionStatusDescription": "Ţĥē ḿōśţ ŕēćēńţ śŵēēƥ ōũţćōḿē àńď ţĥē ćũŕŕēńţ ŕēţàĩńēď-ŕōŵ ćōũńţś ƒōŕ ēàćĥ ţĥŕēśĥōĺďēď ţàƀĺē.", + "retentionNeverSweptMessage": "Ńō śŵēēƥ ĥàś ŕũń ŷēţ śĩńćē ţĥē ƀàćķēńď ĺàśţ śţàŕţēď.", + "retentionSweepStartedAtLabel": "Śţàŕţēď", + "retentionSweepFinishedAtLabel": "Ƒĩńĩśĥēď", + "retentionDeletedLabel": "Ďēĺēţēď", + "retentionWouldDeleteLabel": "Ŵōũĺď ďēĺēţē (ƥŕēvĩēŵ)", + "retentionBacklogLabel": "Ɓàćķĺōĝ: ḿōŕē ēĺĩĝĩƀĺē ŕōŵś ŕēḿàĩń ƥàśţ ţĥĩś śŵēēƥ'ś ƀàţćĥ ĺĩḿĩţ", + "retentionTableErrorLabel": "Ēŕŕōŕ", + "retentionSkipCountLabel": "Śŵēēƥś śķĩƥƥēď", + "retentionSkipCountHelp": "Ćōũńţś śŵēēƥś ţĥàţ ŵēŕē ďũē ƀũţ ďĩď ńōţ ŕũń, ḿōśţ ōƒţēń ƀēćàũśē àńōţĥēŕ Ķàńďēv ƥŕōćēśś ĥēĺď ţĥē ŕēţēńţĩōń ĺōćķ àţ ţĥē śàḿē ţĩḿē.", + "retentionLastSkipAtLabel": "Ĺàśţ śķĩƥƥēď", + "retentionRetainedCountsTitle": "Ŕēţàĩńēď ŕōŵś", + "retentionCensusNotComputed": "Ńōţ ŷēţ ḿēàśũŕēď", + "retentionCensusStale": "Śţàĺē: ĺàśţ ḿēàśũŕēḿēńţ ƒàĩĺēď, śĥōŵĩńĝ ţĥē ĺàśţ śũććēśśƒũĺ ćōũńţ", + "retentionCensusAsOfLabel": "Àś ōƒ", + "retentionUnknownStatusesLabel": "Ŕōŵś ŵĩţĥ àń ũńŕēćōĝńĩźēď śţàţũś (ńōţ ćōũńţēď àś ĥĩśţōŕŷ ōŕ ĺĩvē śţàţē)", + "retentionTopRoutineShareLabel": "Ĺàŕĝēśţ śĩńĝĺē ŕōũţĩńē'ś śĥàŕē ōƒ ŕēţàĩńēď ŕōŵś", "backendReloadRequiredTitle": "Ŕēĺōàď ŕēqũĩŕēď", "backendReloadRequiredBody": "Ķàńďēv ŕēśţàŕţēď. Ŕēĺōàď ţĥĩś ƥàĝē ţō ćōńţĩńũē. Ŕēĺōàďĩńĝ ďĩśćàŕďś ũńśàvēď ćĥàńĝēś.", "backendReloadRequiredAction": "Ŕēĺōàď ƥàĝē", diff --git a/apps/web/src/locales/pt-pt/system.json b/apps/web/src/locales/pt-pt/system.json index dc9122adfaf..097df6de1ca 100644 --- a/apps/web/src/locales/pt-pt/system.json +++ b/apps/web/src/locales/pt-pt/system.json @@ -202,8 +202,52 @@ "navFeatureToggles": "Interruptores de funcionalidades", "navLicenses": "Licenças", "navLogs": "Registos", + "navRetention": "Retenção do histórico de execuções", "navUpdates": "Atualizações", "navUsers": "Utilizadores", + "retentionPageDescription": "Limita durante quanto tempo o histórico de execuções do Office (office_routine_runs, runs e as respetivas tabelas satélite) permanece na base de dados.", + "retentionAdminOnly": "Só um administrador pode alterar as definições de retenção.", + "retentionLoadFailed": "Não foi possível carregar as definições de retenção.", + "retentionSaveFailed": "Não foi possível guardar as definições de retenção.", + "retentionPolicyTitle": "Política de retenção", + "retentionPolicyDescription": "Uma limpeza agendada, separada do ciclo do Office de 5 segundos, elimina o histórico de execuções concluído assim que ultrapassa a sua janela de retenção. As linhas que representam trabalho em curso nunca são eliminadas, independentemente destas definições.", + "retentionEnabledLabel": "Eliminar histórico de execuções elegível", + "retentionEnabledDescription": "Quando desativado, o Kandev nunca elimina linhas de office_routine_runs ou runs. As contagens de linhas retidas abaixo continuam a ser atualizadas para poder ver a acumulação de atrasos antes de reativar a eliminação.", + "retentionSweepIntervalLabel": "Intervalo de limpeza (horas)", + "retentionSweepIntervalHelp": "Frequência com que a limpeza é executada, entre 1 e 168 horas (1 semana).", + "retentionBatchLimitLabel": "Linhas eliminadas por limpeza", + "retentionBatchLimitHelp": "Número máximo de linhas eliminadas por tabela numa limpeza, entre 100 e 100 000. Reduza este valor se a limpeza competir com outra carga na base de dados; um atraso grande é eliminado ao longo de várias limpezas em vez de uma só.", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "Registos de execução de rotinas: uma linha por cada execução de rotina concluída ou falhada.", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "Registos de execução de tarefas: uma linha por cada execução concluída, falhada ou cancelada.", + "retentionRunEventsSectionTitle": "run_events e tabelas satélite", + "retentionRunEventsSectionDescription": "As linhas de run_events, office_run_route_attempts e office_run_skills são eliminadas em conjunto com a execução que as possui; não têm janela nem limite mínimo próprios.", + "retentionWindowDaysLabel": "Janela de retenção (dias)", + "retentionWindowDaysHelp": "As linhas concluídas com mais deste número de dias tornam-se elegíveis para eliminação, entre 1 e 3650 dias.", + "retentionFloorPerOwnerLabel": "Mínimo mantido por proprietário", + "retentionFloorPerOwnerHelp": "Mantém sempre pelo menos este número de linhas concluídas mais recentes de cada proprietário, mesmo além da janela de retenção, entre 0 e 10 000. Defina 0 para desativar este limite mínimo.", + "retentionWarnRowsLabel": "Avisar acima deste número de linhas retidas", + "retentionWarnRowsHelp": "Gera um aviso de saúde quando as linhas retidas ultrapassam este número. Defina 0 para desativar o aviso.", + "retentionRunEventsWarnRowsHelp": "Gera um aviso de saúde quando o total de linhas de run_events retidas ultrapassa este número. Defina 0 para desativar o aviso.", + "retentionStatusTitle": "Estado da retenção", + "retentionStatusDescription": "O resultado da limpeza mais recente e as contagens atuais de linhas retidas para cada tabela com limiar definido.", + "retentionNeverSweptMessage": "Ainda não foi executada nenhuma limpeza desde o último arranque do backend.", + "retentionSweepStartedAtLabel": "Iniciada", + "retentionSweepFinishedAtLabel": "Concluída", + "retentionDeletedLabel": "Eliminadas", + "retentionWouldDeleteLabel": "Seriam eliminadas (pré-visualização)", + "retentionBacklogLabel": "Atraso: ainda há mais linhas elegíveis além do limite desta limpeza", + "retentionTableErrorLabel": "Erro", + "retentionSkipCountLabel": "Limpezas ignoradas", + "retentionSkipCountHelp": "Conta as limpezas que deviam ter sido executadas mas não o foram, geralmente porque outro processo do Kandev detinha o bloqueio de retenção ao mesmo tempo.", + "retentionLastSkipAtLabel": "Última ignorada em", + "retentionRetainedCountsTitle": "Linhas retidas", + "retentionCensusNotComputed": "Ainda não medido", + "retentionCensusStale": "Desatualizado: a última medição falhou, a mostrar a última contagem bem-sucedida", + "retentionCensusAsOfLabel": "Referente a", + "retentionUnknownStatusesLabel": "Linhas com um estado não reconhecido (não contadas como histórico nem como estado ativo)", + "retentionTopRoutineShareLabel": "Proporção de linhas retidas pertencentes à maior rotina individual", "backendReloadRequiredTitle": "É necessário recarregar", "backendReloadRequiredBody": "O Kandev foi reiniciado. Recarregue esta página para continuar. O recarregamento elimina as alterações não guardadas.", "backendReloadRequiredAction": "Recarregar página", diff --git a/apps/web/src/locales/zh-cn/system.json b/apps/web/src/locales/zh-cn/system.json index 333cbe57e8c..1361e839d65 100644 --- a/apps/web/src/locales/zh-cn/system.json +++ b/apps/web/src/locales/zh-cn/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能开关", "navLicenses": "许可证", "navLogs": "日志", + "navRetention": "运行历史保留", "navUpdates": "更新", "navUsers": "用户", + "retentionPageDescription": "限制 Office 运行历史(office_routine_runs、runs 及其附属表)在数据库中保留的时长。", + "retentionAdminOnly": "只有管理员才能更改保留设置。", + "retentionLoadFailed": "无法加载保留设置。", + "retentionSaveFailed": "无法保存保留设置。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一个与 5 秒 Office 心跳独立的定时清理任务,会在已完成的运行历史超出其保留窗口后将其删除。代表当前进行中工作的行永远不会被删除,与这些设置无关。", + "retentionEnabledLabel": "删除符合条件的运行历史", + "retentionEnabledDescription": "关闭时,Kandev 永远不会删除 office_routine_runs 或 runs 中的行。下方的保留行数仍会持续更新,方便你在重新启用删除前查看积压情况。", + "retentionSweepIntervalLabel": "清理间隔(小时)", + "retentionSweepIntervalHelp": "清理任务运行的频率,介于 1 到 168 小时(1 周)之间。", + "retentionBatchLimitLabel": "每次清理删除的行数", + "retentionBatchLimitHelp": "每次清理中每张表最多删除的行数,介于 100 到 100000 之间。如果清理与其他数据库负载相互争用,可以调低此值;较大的积压会分多次清理逐步处理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任务执行记录:每完成或失败一次例行任务运行对应一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任务运行记录:每完成、失败或取消一次运行对应一行。", + "retentionRunEventsSectionTitle": "run_events 及附属表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行会随其所属的运行一起删除;它们没有自己的窗口或下限设置。", + "retentionWindowDaysLabel": "保留窗口(天)", + "retentionWindowDaysHelp": "已完成且超过此天数的行将符合删除条件,介于 1 到 3650 天之间。", + "retentionFloorPerOwnerLabel": "每个所有者的最少保留数", + "retentionFloorPerOwnerHelp": "始终至少保留每个所有者最近完成的这么多行,即使超出保留窗口,介于 0 到 10000 之间。设为 0 可停用此下限。", + "retentionWarnRowsLabel": "超过此保留行数时发出警告", + "retentionWarnRowsHelp": "当保留行数超过此数值时触发健康警告。设为 0 可停用该警告。", + "retentionRunEventsWarnRowsHelp": "当保留的 run_events 总行数超过此数值时触发健康警告。设为 0 可停用该警告。", + "retentionStatusTitle": "保留状态", + "retentionStatusDescription": "最近一次清理的结果,以及每张设有阈值的表当前的保留行数。", + "retentionNeverSweptMessage": "自后端上次启动以来,尚未运行过清理任务。", + "retentionSweepStartedAtLabel": "开始于", + "retentionSweepFinishedAtLabel": "结束于", + "retentionDeletedLabel": "已删除", + "retentionWouldDeleteLabel": "预计删除(预览)", + "retentionBacklogLabel": "积压:超出本次清理批量上限的符合条件的行仍然存在", + "retentionTableErrorLabel": "错误", + "retentionSkipCountLabel": "已跳过的清理次数", + "retentionSkipCountHelp": "统计原本应该运行但未运行的清理次数,通常是因为同一时间有另一个 Kandev 进程持有保留锁。", + "retentionLastSkipAtLabel": "上次跳过于", + "retentionRetainedCountsTitle": "保留的行数", + "retentionCensusNotComputed": "尚未统计", + "retentionCensusStale": "已过期:上次统计失败,显示的是上一次成功的计数", + "retentionCensusAsOfLabel": "统计时间", + "retentionUnknownStatusesLabel": "状态无法识别的行(既不计入历史,也不计入当前状态)", + "retentionTopRoutineShareLabel": "占保留行数比例最高的单个例行任务", "backendReloadRequiredTitle": "需要重新加载", "backendReloadRequiredBody": "Kandev 已重启。请重新加载此页面以继续。重新加载会丢弃未保存的更改。", "backendReloadRequiredAction": "重新加载页面", diff --git a/apps/web/src/locales/zh-hk/system.json b/apps/web/src/locales/zh-hk/system.json index 38563613bbe..3f65e70e084 100644 --- a/apps/web/src/locales/zh-hk/system.json +++ b/apps/web/src/locales/zh-hk/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能開關", "navLicenses": "許可證", "navLogs": "日誌", + "navRetention": "執行歷史保留", "navUpdates": "更新", "navUsers": "用戶", + "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在數據庫中保留的時長。", + "retentionAdminOnly": "只有管理員才能更改保留設定。", + "retentionLoadFailed": "無法載入保留設定。", + "retentionSaveFailed": "無法儲存保留設定。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留窗口後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。", + "retentionEnabledLabel": "刪除符合條件的執行歷史", + "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啓用刪除前查看積壓情況。", + "retentionSweepIntervalLabel": "清理間隔(小時)", + "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。", + "retentionBatchLimitLabel": "每次清理刪除的行數", + "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他數據庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。", + "retentionRunEventsSectionTitle": "run_events 及附屬表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的窗口或下限設定。", + "retentionWindowDaysLabel": "保留窗口(天)", + "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。", + "retentionFloorPerOwnerLabel": "每個所有者的最少保留數", + "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留窗口,介於 0 到 10000 之間。設為 0 可停用此下限。", + "retentionWarnRowsLabel": "超過此保留行數時發出警告", + "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionStatusTitle": "保留狀態", + "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。", + "retentionNeverSweptMessage": "自後端上次啓動以來,尚未執行過清理任務。", + "retentionSweepStartedAtLabel": "開始於", + "retentionSweepFinishedAtLabel": "結束於", + "retentionDeletedLabel": "已刪除", + "retentionWouldDeleteLabel": "預計刪除(預覽)", + "retentionBacklogLabel": "積壓:超出本次清理批量上限的符合條件的行仍然存在", + "retentionTableErrorLabel": "錯誤", + "retentionSkipCountLabel": "已跳過的清理次數", + "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 程序持有保留鎖。", + "retentionLastSkipAtLabel": "上次跳過於", + "retentionRetainedCountsTitle": "保留的行數", + "retentionCensusNotComputed": "尚未統計", + "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數", + "retentionCensusAsOfLabel": "統計時間", + "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)", + "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務", "backendReloadRequiredTitle": "需要重新載入", "backendReloadRequiredBody": "Kandev 已重啓。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。", "backendReloadRequiredAction": "重新載入頁面", diff --git a/apps/web/src/locales/zh-tw/system.json b/apps/web/src/locales/zh-tw/system.json index 8aec9747198..192eec0cddb 100644 --- a/apps/web/src/locales/zh-tw/system.json +++ b/apps/web/src/locales/zh-tw/system.json @@ -204,8 +204,52 @@ "navFeatureToggles": "功能開關", "navLicenses": "許可證", "navLogs": "日誌", + "navRetention": "執行歷史保留", "navUpdates": "更新", "navUsers": "使用者", + "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在資料庫中保留的時長。", + "retentionAdminOnly": "只有管理員才能更改保留設定。", + "retentionLoadFailed": "無法載入保留設定。", + "retentionSaveFailed": "無法儲存保留設定。", + "retentionPolicyTitle": "保留策略", + "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留視窗後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。", + "retentionEnabledLabel": "刪除符合條件的執行歷史", + "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啟用刪除前檢視積壓情況。", + "retentionSweepIntervalLabel": "清理間隔(小時)", + "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。", + "retentionBatchLimitLabel": "每次清理刪除的行數", + "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他資料庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。", + "retentionRoutineRunsSectionTitle": "office_routine_runs", + "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。", + "retentionRunsSectionTitle": "runs", + "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。", + "retentionRunEventsSectionTitle": "run_events 及附屬表", + "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的視窗或下限設定。", + "retentionWindowDaysLabel": "保留視窗(天)", + "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。", + "retentionFloorPerOwnerLabel": "每個所有者的最少保留數", + "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留視窗,介於 0 到 10000 之間。設為 0 可停用此下限。", + "retentionWarnRowsLabel": "超過此保留行數時發出警告", + "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。", + "retentionStatusTitle": "保留狀態", + "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。", + "retentionNeverSweptMessage": "自後端上次啟動以來,尚未執行過清理任務。", + "retentionSweepStartedAtLabel": "開始於", + "retentionSweepFinishedAtLabel": "結束於", + "retentionDeletedLabel": "已刪除", + "retentionWouldDeleteLabel": "預計刪除(預覽)", + "retentionBacklogLabel": "積壓:超出本次清理批次上限的符合條件的行仍然存在", + "retentionTableErrorLabel": "錯誤", + "retentionSkipCountLabel": "已跳過的清理次數", + "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 處理程序持有保留鎖。", + "retentionLastSkipAtLabel": "上次跳過於", + "retentionRetainedCountsTitle": "保留的行數", + "retentionCensusNotComputed": "尚未統計", + "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數", + "retentionCensusAsOfLabel": "統計時間", + "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)", + "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務", "backendReloadRequiredTitle": "需要重新載入", "backendReloadRequiredBody": "Kandev 已重啟。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。", "backendReloadRequiredAction": "重新載入頁面", diff --git a/docs/plans/run-history-retention/plan.md b/docs/plans/run-history-retention/plan.md new file mode 100644 index 00000000000..36b58620542 --- /dev/null +++ b/docs/plans/run-history-retention/plan.md @@ -0,0 +1,64 @@ +--- +created: 2026-09-09 +status: complete +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +system_design: + - ../../specs/office/system-design/run-history-retention.md + - ../../specs/office/system-design/run-history-retention-operations.md +legacy_specs: [] +--- + +# Implementation Plan: Office Run History Retention + +## Overview + +Office writes `office_routine_runs` and `run_events` rows on every routine firing and +run lifecycle transition, and nothing ever deletes them on a schedule. A `*/5 * * * *` +routine produces roughly 105,000 `office_routine_runs` rows a year; one reference +install had already accumulated 323 consecutive `coalesced` rows from a routine that +was doing nothing useful. This plan adds one scheduled sweep, on its own interval +(never the 5s Office tick), that bounds `office_routine_runs`, `runs`, and their +satellites (`run_events`, `office_run_route_attempts`, `office_run_skills`) by age, +with a per-owner floor, identical behavior on SQLite and PostgreSQL, and an operator +surface (Settings > System > Data & Logs) that reports policy, counts, previews, and +warnings before and while rows are deleted. + +The two halves of the contract are split the same way the specs are split: the sweep +itself, its eligibility rules, and engine parity are +[run history retention](../../specs/office/system-design/run-history-retention.md) +(REQ-001, REQ-002, REQ-005); the settings record, preview marker, health warnings, and +System page surface are +[run history retention operations](../../specs/office/system-design/run-history-retention-operations.md) +(REQ-003, REQ-004). + +## Scope + +### In scope + +- A `internal/office/retention` package: settings store, eligibility/count/delete + queries shared by preview and delete, a session-scoped PostgreSQL advisory lock with + a SQLite single-process equivalent, a scheduler goroutine on its own interval, and an + HTTP handler for `GET`/`PUT /api/v1/system/retention`. +- Status-only classification of history vs. live state for both `office_routine_runs` + and `runs`, a per-owner floor, oldest-first chunked batch deletion, and satellite rows + deleted in the same transaction as their parent run. +- A per-table preview (report, delete nothing) on each table's first evaluation, and + `health.Issue` warnings before the cap and on sweep/count failure. +- A `RetentionSettingsCard` on Settings > System > Data & Logs showing policy, retained + counts, preview state, last sweep, and backlog/error warnings. +- Expression indexes serving the sweep's filter/order on both engines. + +### Out of scope + +- Filesystem/container cleanup (owned by storage maintenance). +- Routine or workspace deletion (already deletes runs structurally; unaffected). +- Any change to run lifecycle, routine dispatch, or task/session/checkout data. + +## Tasks + +- [x] [Task 01: Bound run history with a scheduled retention sweep](task-01-bound-run-history-with-retention-sweep.md) diff --git a/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md new file mode 100644 index 00000000000..7d7c2c15465 --- /dev/null +++ b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md @@ -0,0 +1,183 @@ +--- +id: "01-bound-run-history-with-retention-sweep" +title: "Bound Office run history with a scheduled retention sweep" +status: done +wave: 1 +depends_on: [] +plan: "plan.md" +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +acceptance_criteria: + - AC-OFFICE-RUN-HISTORY-RETENTION-001.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-001.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.11 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.12 + - AC-OFFICE-RUN-HISTORY-RETENTION-002.13 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.10 + - AC-OFFICE-RUN-HISTORY-RETENTION-003.11 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.5 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.6 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.7 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.8 + - AC-OFFICE-RUN-HISTORY-RETENTION-004.9 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.1 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.2 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.3 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.4 + - AC-OFFICE-RUN-HISTORY-RETENTION-005.5 +system_design: + - ../../specs/office/system-design/run-history-retention.md + - ../../specs/office/system-design/run-history-retention-operations.md +--- + +# Task 01: Bound Office Run History with a Scheduled Retention Sweep + +## Summary + +Add `internal/office/retention`: a settings-backed sweep that ages out +`office_routine_runs` and `runs` history (plus their satellites) on its own +interval, a per-owner floor, status-only live/history classification, a preview +pass before first deletion, engine-identical behavior on SQLite and PostgreSQL, +and a `GET`/`PUT /api/v1/system/retention` operator surface with a matching +Settings > System > Data & Logs card. + +## In scope + +- Eligibility, count, and delete queries shared by preview and the real sweep, + keyed on `COALESCE(completed_at, created_at)` / `COALESCE(finished_at, + requested_at)`, oldest-first, chunked across statements. +- A session-scoped PostgreSQL advisory lock (with a SQLite single-process + equivalent) so only one backend sweeps at a time. +- A per-owner floor (newest N rows retained regardless of age) applied to + count, preview, and delete identically, re-asserted in the DELETE statement. +- Satellite deletion (`run_events`, `office_run_route_attempts`, + `office_run_skills`) in the same transaction as their parent run. +- `health.Issue` warnings for approaching the backlog cap and for count/sweep + failure; a one-time preview per table before any row is deleted. +- `RetentionSettingsCard` on Settings > System > Data & Logs: policy, retained + counts, preview/backlog state, last sweep, errors. +- Expression indexes serving the sweep's filter/order on both engines. + +## Out of scope + +- Filesystem/container artifact cleanup (storage maintenance owns that). +- Routine/workspace deletion cascades (already delete runs structurally). +- Any run lifecycle, routing, or task/session/checkout behavior change. + +## Acceptance + +- History rows (terminal `office_routine_runs`/`runs` statuses) older than the + configured window are deleted in batches; live-state rows (`received`, + `task_created`, `queued`, `claimed`) are never age-pruned; an unrecognized + status fails safe as live state and warns. +- The newest `floor` rows per owner survive regardless of age. +- Each table's first sweep is a preview: it reports would-delete counts and + deletes nothing; a later sweep performs real deletion. +- Behavior, including the advisory lock and batch chunking, is identical on + SQLite and PostgreSQL. +- `GET`/`PUT /api/v1/system/retention` read/write policy and report current + counts, preview state, last sweep outcome, and backlog/error warnings. + +## Verification + +```bash +cd apps/backend && go test -race -count=1 ./internal/office/retention/... ./internal/office/repository/sqlite/... +cd apps/backend && KANDEV_TEST_POSTGRES_DSN= go test -race -count=1 -v ./internal/office/retention/... +cd apps/backend && golangci-lint run ./internal/office/retention/... +cd apps/web && pnpm run typecheck +cd apps && pnpm --filter @kandev/web test -- --run retention-settings-card system-route-copy +cd apps/web && pnpm e2e:run -- --project chromium --grep "System retention settings" +``` + +## Files likely touched + +- `apps/backend/internal/office/retention/*.go` +- `apps/backend/internal/office/repository/sqlite/base_migrations.go`, + `retention_indexes_test.go`, `retention_indexes_postgres_test.go` +- `apps/backend/internal/backendapp/*` (scheduler/handler wiring) +- `apps/web/src/**/retention-settings-card.tsx` and its tests +- `docs/specs/office/requirements/run-history-retention*.md`, + `docs/specs/office/system-design/run-history-retention*.md` +- `docs/public/operations.md` + +## Dependencies + +None. + +## Risks + +- A dialect-sensitive query bug that only reproduces on PostgreSQL (mitigated + by an environment-gated Postgres suite covering the advisory lock, the + EvalPlanQual TOCTOU window, and batch chunking). +- A too-aggressive window or floor deleting rows an operator still needed + (mitigated by the default 30-day window, the per-owner floor, and the + preview-before-delete behavior). + +## Parallelism + +`sequential` + +## Inputs + +- `REQ-OFFICE-RUN-HISTORY-RETENTION-001` through `-005`. +- [run history retention](../../specs/office/system-design/run-history-retention.md) + and + [run history retention operations](../../specs/office/system-design/run-history-retention-operations.md). +- The reference install's 323 consecutive `coalesced` routine-run rows (28 + days, one bricked routine) cited in the requirements' Overview. + +## Results + +- Implemented `internal/office/retention` (settings store, `Sweeper`, + `Scheduler`, `CensusTracker`, PostgreSQL advisory lock with SQLite + equivalent, `Handler` for `GET`/`PUT /api/v1/system/retention`) plus the + `RetentionSettingsCard` on Settings > System > Data & Logs. +- Full backend gauntlet green: `go build ./...`, `go vet`, `gofmt -l`, + `go test -race ./internal/office/retention/...` (SQLite), the + PostgreSQL-gated suite against a real scratch instance (125 tests, 0 + skipped, all PASS, covering the advisory lock, the EvalPlanQual + concurrent-resurrection TOCTOU regression, and chunked batch deletes), + `golangci-lint run ./...` (0 issues). +- Frontend: `pnpm run typecheck`, the retention card and System route-copy + Vitest suites, and a scoped Playwright run (`--grep "System retention + settings"`, 2 passed). +- Delivered as PR [#3566](https://github.com/kdlbs/kandev/pull/3566) + ("feat(office): bound run history growth with a scheduled retention + sweep"), through four Build rounds and four Review rounds; remaining + non-blocking test-rigor gaps and CI-wiring follow-ups are tracked on the + linked follow-up card rather than blocking this PR. diff --git a/docs/public/operations.md b/docs/public/operations.md index 7ce3be85abd..58d19172a06 100644 --- a/docs/public/operations.md +++ b/docs/public/operations.md @@ -348,6 +348,34 @@ the Kandev service first provides the clearest maintenance boundary. +## Office run history retention + +Open **Settings > System > Data & Logs** to manage automatic deletion of old +Office run history. Deletion is enabled by default. The first sweep starts five +minutes after the backend starts or after you enable deletion. + +The first sweep for each history table is a preview. It reports the rows that +would be deleted and removes no rows. A later sweep can delete eligible rows. +Deletion is permanent. Back up the database before you enable deletion if you +need to keep old history outside the configured window. + +The retention window controls the age of rows that can be deleted. The minimum +kept per owner control keeps the newest rows for each routine or agent, even +when those rows are older than the window. A floor of zero removes this extra +protection. Run event, route attempt, and skill rows are deleted with their +parent run. + +The page shows the current policy, retained row counts, preview results, the +last sweep, and any backlog or errors. Counts continue to update when deletion +is disabled, so you can monitor growth before you enable it again. + +To disable automatic deletion: + +1. Open **Settings > System > Data & Logs**. +2. Clear **Delete eligible run history**. +3. Select **Save changes**. +4. Check the retention status. It must show that deletion is disabled. + ## Database operation > **Single-owner rule:** SQLite uses one writer connection in WAL mode; only one Kandev backend should own the file. **Factory reset** is destructive and removes managed data after creating a pre-reset backup. diff --git a/docs/specs/office/requirements/run-history-retention-operations.md b/docs/specs/office/requirements/run-history-retention-operations.md new file mode 100644 index 00000000000..394f70290e9 --- /dev/null +++ b/docs/specs/office/requirements/run-history-retention-operations.md @@ -0,0 +1,214 @@ +--- +status: draft +system: office +created: 2026-09-09 +owners: + - kandev +--- + +# Office Run History Retention Operations Requirements + +## Overview + +[Run history retention](run-history-retention.md) defines which Office run +history rows may be deleted and how the sweep that deletes them behaves. This +document defines the other half of that contract: what an operator is told +before the first row is removed, what they are told while the tables grow, what +they can configure, and what they can read about the last sweep. + +The split follows the ownership boundary the retention contract already draws. +Office owns the rows and the deletion policy because they are Office primitives. +The System pages own the operator surface those decisions are reported through, +and that surface is a separate contract with a separate consumer: an operator +reading a settings page and a health card, rather than a scheduler deleting +rows. Splitting here keeps each document inside its size limit without cutting +either contract. + +The requirement IDs continue the retention capability's sequence rather than +starting a new one, because these are the same capability's requirements viewed +from the operator's side. + +## Terminology + +Terms are defined once, in +[run history retention](run-history-retention.md#terminology), and used here +with the same meaning. The ones this document leans on most: + +- **Retention sweep**, **preview**, and **retained count**. +- **Swept tables** (`office_routine_runs`, `runs`), **reported tables** (those + two plus the three run satellites), and **thresholded tables** + (`office_routine_runs`, `runs`, `run_events`). "Per table" always names one of + these three sets, never an unqualified "table". + +## Requirements + +### REQ-OFFICE-RUN-HISTORY-RETENTION-003: Warn before deleting, and warn while growing + +**Intent:** An operator learns what retention is about to remove before it +removes anything, and learns that history is growing past a threshold while +there is still time to widen the window or fix the routine. + +As an operator upgrading an install with a year of run history, I want to be +told what the first sweep would delete before it deletes it, so that I can widen +the retention window first if that history matters to me. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.1:** The first evaluation of each swept table + on a database shall be a preview for that table: it evaluates the policy, + reports the number of rows it would delete from that table, and deletes + nothing from it. A table that has not completed a preview never deletes, + regardless of what any other table has done. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.2:** When a preview sweep would delete + at least one row, the system shall emit an operator-visible warning naming + each table, its would-delete count, and the configured retention window, and + stating that deletion begins at the next scheduled sweep. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.3:** When a preview sweep would delete + no rows, the system shall not emit a warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.4:** The system shall run a preview at most + once per swept table per database. A table's preview shall be recorded only + when that table's preview evaluation completed successfully; a table that + failed during a sweep in which it was being previewed shall be previewed again + on the next sweep rather than deleting. Disabling and re-enabling retention, + restarting the backend, or changing any retention setting shall not produce a + second preview for a table that has already completed one. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.5:** When a thresholded table's retained + count exceeds that table's configured warning threshold, the system shall emit + an operator-visible warning naming the table, its retained count, and the + threshold. For `office_routine_runs`, the warning shall also name the routine + holding the largest share of retained rows and that share, where the share is + that routine's retained rows as a proportion of the table's retained count. + When two routines hold an equal largest share, the lower routine identifier + shall be named. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.6:** When a table reports remaining + backlog after a sweep, the system shall emit an operator-visible warning + naming the table, stating that retention is behind, and reporting the number + of rows deleted in that sweep. A table that was previewed in that sweep shall + not produce this warning however many rows were eligible, because a preview + deletes nothing by design and retention is therefore not behind; the preview + warning of AC-OFFICE-RUN-HISTORY-RETENTION-003.2 reports its eligible count + instead. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.7:** When retention is disabled and a + table's retained count exceeds its warning threshold, the system shall emit + the same threshold warning and shall additionally state that retention is + disabled, so a silent unbounded table is distinguishable from one being + actively managed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.8:** Every warning in this requirement + shall be emitted through a channel available in a production build, and shall + not be observable only through the debug metrics endpoint. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.9:** A preview sweep's would-delete + counts shall be the full eligible count per table, not capped by the batch + limit, so an operator is told the real size of what is about to be removed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.10:** When a swept table's recorded preview + state cannot be read or parsed, the system shall treat that table as not yet + previewed and shall emit an operator-visible warning naming the table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-003.11:** Retained counts for the thresholded + tables shall be produced by a count evaluation that does not require a sweep + to have run, so the counts and the threshold warnings required by AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and + AC-OFFICE-RUN-HISTORY-RETENTION-004.8 are available on a fresh install and while retention is disabled. The + counts shall be refreshed on the sweep interval and reused between refreshes + rather than recomputed for each operator page load or health poll. The first + count evaluation shall run at startup, before the delay of + AC-OFFICE-RUN-HISTORY-RETENTION-002.10 arms the first sweep, so an operator is + not shown countless tables for a sweep interval after every restart. When a + count evaluation fails, the system shall keep the counts from the last + successful evaluation, report them as stale together with the time they were + produced, and emit an operator-visible warning. A failed count shall not + suppress the threshold warnings derived from the last successful counts, and + shall not fail the sweep. Until the first evaluation has completed, and when it + fails with no earlier successful evaluation to fall back on, the system shall + report the counts as not yet computed and shall not render them as zero, on the + same grounds as AC-OFFICE-RUN-HISTORY-RETENTION-004.7: a count nobody has taken + must not read as a table that is empty. Success and failure shall be tracked + per thresholded table, so one table's failed count neither discards nor marks + stale another table's successful one. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-004: Operator configuration and sweep visibility + +**Intent:** Retention is configurable, its values are validated, and the result +of the most recent sweep is readable by an operator without reading logs. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.1:** An operator can read and change + whether retention is enabled, the retention window per in-scope table, the + sweep interval, the retention floor per in-scope table, the batch limit, and + the warning threshold per in-scope table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.2:** Retention shall be enabled by + default. Default values shall be: retention window 30 days for both + `office_routine_runs` and `runs`; sweep interval 6 hours; retention floor 50 + rows per owner for both tables; batch limit 5,000 rows per table per sweep; + warning threshold 25,000 rows for `office_routine_runs`, 25,000 for `runs`, + and 250,000 for `run_events`. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.3:** The system shall reject a setting + outside its permitted range with an error naming the field, and shall leave + the stored settings unchanged. Permitted ranges are: retention window 1 to + 3,650 days; sweep interval 1 to 168 hours; retention floor 0 to 10,000; batch + limit 100 to 100,000; warning threshold 0 or greater, where 0 disables that + table's threshold warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.4:** When stored retention settings + cannot be read or parsed, the system shall use the documented defaults, emit + an operator-visible warning, and continue. Unreadable settings shall not + disable retention silently and shall not fail startup. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.5:** A settings change shall take effect + without a backend restart, and shall apply from the next scheduled sweep + rather than interrupting a sweep in progress. Two concurrent writes resolve + last-writer-wins; repeating an identical write changes nothing and returns the + same normalized document. A sweep shall read the stored settings at its start + rather than relying on a value cached when this process last observed a change, + so that where several backend processes share one database a process that did + not serve the write still sweeps under the new settings rather than the + replaced ones. When that read fails or the stored settings cannot be parsed, + the sweep shall be skipped and recorded as skipped rather than run against the + documented defaults, because a default window is shorter than a window an + operator has widened and sweeping under it would delete the history they + configured the system to keep. This does not change + AC-OFFICE-RUN-HISTORY-RETENTION-004.4, which governs reading settings for + reporting and startup, where using the defaults deletes nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.6:** An operator can read, on the System + pages, the outcome of the most recent completed sweep: when it ran, which + swept tables it previewed, the rows deleted per reported table, the rows it + would have deleted per previewed table, the retained count per thresholded + table, whether any table has remaining backlog, and any table that failed. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.7:** When no sweep has run since the + backend started, the surface in AC-OFFICE-RUN-HISTORY-RETENTION-004.6 shall + say so explicitly rather than render an empty or zeroed result that reads as a + sweep that deleted nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.8:** The sweep-result surface shall be + readable while retention is disabled, reporting retained counts and the + disabled state. +- **AC-OFFICE-RUN-HISTORY-RETENTION-004.9:** A settings write shall replace the whole + settings document. A field the caller omits shall take its documented default + rather than its previously stored value, so the same request always produces + the same stored document. An unrecognized field, a numeric field whose + value is not a whole number, a field present with a JSON `null`, and a field + whose value is not of its documented type, shall each be rejected with an error + naming the field, leaving the stored settings unchanged. `null` shall be + rejected rather than treated as omission: omission means the documented + default, so reading `null` the same way would let a client silently replace a + configured retention window with a shorter default and destroy history the + operator meant to keep. + +## Out of scope + +- **A persisted history of retention sweeps.** The last sweep's result is held + in memory and reported alongside the durable warnings. A growing table + recording the work of the job that stops tables growing is the same defect in + a new place. +- **Per-workspace or per-routine retention overrides.** Settings are + instance-wide. The retention floor is already per owner, which covers what a + per-routine override would most often be used for. +- **Archival or export before deletion.** Deleted history is gone. An install + needing it kept sets a longer window or takes a backup. +- **Anything the deletion policy owns.** Eligibility, ordering, batching, + atomicity, and engine parity are specified in + [run history retention](run-history-retention.md) and are not restated here. + +## Prior art + +The receipts for both prior-art legs are recorded once, in +[run history retention](run-history-retention.md#prior-art). The finding that +bears on this document specifically: neither surveyed product previews before a +policy's first deletion, nor warns ahead of the window. Those two behaviors are +where this capability goes further, and they are the reason this document +exists rather than being a settings page bolted onto a sweep. diff --git a/docs/specs/office/requirements/run-history-retention.md b/docs/specs/office/requirements/run-history-retention.md new file mode 100644 index 00000000000..535a1bf9d04 --- /dev/null +++ b/docs/specs/office/requirements/run-history-retention.md @@ -0,0 +1,324 @@ +--- +status: draft +system: office +created: 2026-09-09 +owners: + - kandev +--- + +# Office Run History Retention Requirements + +## Overview + +Office writes two families of history rows that nothing removes on a schedule. +`office_routine_runs` records one row per routine firing. `run_events` records +the timeline of every Office run, append-only, alongside the `runs` queue row +and its per-run satellites. The only deletions today are manual and structural: +deleting a routine removes its runs, and deleting a workspace removes everything +belonging to it. An install that never deletes a routine or a workspace grows +forever. + +The growth is driven by a clock, not by usage. A routine on a `*/5 * * * *` +schedule fires 105,120 times a year, and each firing writes a routine-run row +whether or not it does anything. On the reference install, one routine had +accumulated 323 consecutive `coalesced` rows over 28 days while producing no +work at all: that is the curve with the loop broken, and a working loop also +writes the run row, its `run_events` timeline, and its route-attempt rows. + +This document bounds those tables. It defines what is history and may be +deleted, what is live state and must never be deleted on age, when the deletion +runs, and that it behaves identically on both database engines. What an operator +is told before the first row is removed, what they are warned about while the +tables grow, and what they can configure and read is the paired contract in +[run history retention operations](run-history-retention-operations.md). + +Office owns this contract because the rows are defined by Office primitives: the +routine dispatch ledger, the run queue, and the run event timeline. The System +pages own the operator surface it reports through. The filesystem and container +cleanup owned by [storage +maintenance](../../system-page/requirements/storage-maintenance.md) is a +separate capability that never touches database rows. + +## Terminology + +- **Retention sweep** (or **sweep**): one pass that evaluates every in-scope + table against the configured policy and deletes the eligible rows. +- **History row**: a row recording something that already happened, which no + live decision reads. History rows are eligible for deletion. +- **Live-state row**: a row a live decision still reads, regardless of age. + Never eligible for age-based deletion. +- **Recovery-protected run**: a failed `runs` row named by an active + `office_agent_pause_recoveries.failed_run_id`; retention keeps it until the + recovery row is consumed or discarded. +- **Run satellite row**: a row keyed by a `runs` row's identifier and owned by + it: a `run_events`, `office_run_route_attempts`, or `office_run_skills` entry. +- **Retention window**: the age past which a history row becomes eligible, + measured from the row's own completion time. +- **Completion time**: `COALESCE(completed_at, created_at)` for a routine-run + row and `COALESCE(finished_at, requested_at)` for a `runs` row. Both fallback + columns are `NOT NULL`, so completion time is never null. +- **Retention floor**: most-recent history rows kept per owner regardless of + age, so a rarely-firing routine or rarely-woken agent keeps visible history. +- **Batch limit**: the maximum rows one sweep deletes from one table. +- **Preview**: a swept table's first evaluation on a database; it counts what it + would delete from that table and deletes nothing. Tracked per swept table. +- **Retained count**: the number of rows a table currently holds — a property of + the table, not of a sweep, defined whether or not a sweep has ever run and + whether or not retention is enabled. +- **Swept tables**: the two tables retention selects rows from by policy, + `office_routine_runs` and `runs`. +- **Reported tables**: the five tables a sweep can delete rows from and reports + deleted counts for: the two swept tables plus `run_events`, + `office_run_route_attempts`, and `office_run_skills`. +- **Thresholded tables**: the three tables carrying a warning threshold, + `office_routine_runs`, `runs`, and `run_events`. + +## Requirements + +### REQ-OFFICE-RUN-HISTORY-RETENTION-001: History is bounded and live state is not + +**Intent:** Bound `office_routine_runs`, `runs`, and the run satellite tables by +age and by an owner-scoped floor, while guaranteeing that no row another +decision still reads is removed because it is old. + +As an operator running Office continuously, I want old run history removed +automatically, so a scheduled routine does not grow the database without limit. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.1:** A routine-run row is a history row + only when its status is one of `skipped`, `coalesced`, `failed`, `done`, or + `cancelled`. When a routine-run row's status is `received` or `task_created`, + the system shall treat it as a live-state row and shall not delete it on age, + at any age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.2:** A `runs` row is a history row when its + status is `finished`, `failed`, or `cancelled`. `cancelled` is terminal: its + only writer moves a row there from `queued` or `claimed` and stamps the + completion timestamp in the same statement. When a `runs` row's status is + `queued` or `claimed`, the system shall treat it as a live-state row and shall + not delete it on age, at any age, including a run parked for a future routing + retry. A failed run referenced by an active + `office_agent_pause_recoveries.failed_run_id` is also live state for + retention and shall remain until that recovery row is consumed or discarded. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.3:** When a history row's completion time is + older than that table's retention window, the system shall make it eligible + for deletion. Completion time is defined in Terminology. A row's + classification as history shall depend on its status alone and shall not + additionally require `completed_at` or `finished_at` to be set; a terminal row + with an unset stamp is history, dated by its fallback column, and ages out + rather than being retained forever. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.4:** The system shall retain the newest + history rows of each owner up to that table's retention floor even when they + are older than the retention window. The owner is the routine for + `office_routine_runs` and the agent profile for `runs`. Newest is completion + time descending, with the row identifier descending as the tiebreak. + Completion time is never null, so this ordering is total on both engines and + does not depend on either engine's default placement of nulls. The identifier + gives a stable order for equal timestamps and is not claimed to be + chronological. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.5:** When the system deletes a `runs` + row, it shall delete that run's satellite rows in the same database + transaction, and no satellite row shall remain that references a `runs` row + the system has deleted. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.6:** The system shall not delete + `run_events` rows for a run it is not deleting in the same transaction, at any + age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.7:** A routine's status shall not exempt + its history from retention. When a routine is `paused`, its history rows are + evaluated by the same policy as an `active` routine's. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.8:** The system shall not delete, alter, + archive, or cancel a task, a session, a task checkout, or a workspace as part + of a retention sweep. A routine-run row naming a task in `linked_task_id` may + be deleted while that task continues to exist. +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.9:** A retained `coalesced` routine-run + row may name a `coalesced_into_run_id` whose row has already been deleted. + `coalesced_into_run_id` records provenance and is not a referential + constraint; the system shall not delete a coalesced row because its target was + deleted, and shall not retain a target because a coalesced row names it. + +- **AC-OFFICE-RUN-HISTORY-RETENTION-001.10:** When a row in a swept table holds a + status belonging to neither that table's history set nor its live-state set, + the system shall treat it as a live-state row, shall not delete it at any age, + and shall emit an operator-visible warning naming the table and the + unrecognized status, through the channel required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 in [run history + retention operations](run-history-retention-operations.md). The warning shall be + produced by the count evaluation required by + AC-OFFICE-RUN-HISTORY-RETENTION-003.11 rather than by the deletion path, which + cannot observe a status it does not select; it shall therefore be emitted on an + install where retention is disabled and no sweep runs. When one table holds more + than one unrecognized status, the system shall emit a single warning for that + table naming every unrecognized status in ascending lexicographic order with its + row count. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-002: The retention sweep + +**Intent:** Run retention on its own schedule, off the run-claiming hot path, in +bounded batches, with one sweep at a time and no dependence on database cascade +behavior. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.1:** The system shall run retention on a + dedicated schedule whose interval is configurable in hours, and shall not + perform retention work on the Office run-processing tick. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.2:** When a scheduled sweep is due while a + previous sweep is still running, the system shall skip the due sweep rather + than run two concurrently or queue it, and shall record that it was skipped. + Recording a skip shall not replace the reported result of the last completed + sweep. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.3:** A sweep shall delete at most the batch + limit of rows from `office_routine_runs` and at most that many from `runs`. A + preview evaluation shall never be reported as having remaining backlog: it + deletes nothing by design, so "retention is behind" is not true of it however + many rows were eligible. + The limit counts those rows only; every satellite row of a deleted run is + removed regardless of the limit. When more rows were eligible than the limit + allowed, the system shall complete the sweep, report that table as having + remaining backlog, and continue on the next scheduled sweep. Within a table + the sweep shall select its batch in completion-time ascending order, with the + row identifier ascending as the tiebreak, so the oldest eligible rows are + removed first and both engines select the same batch. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.4:** The system shall re-assert every + eligibility condition in the deletion itself, not only when selecting + candidates. A run that returns to `queued` between selection and deletion, + which a scheduled retry does by clearing `finished_at`, shall not be deleted + by that sweep. The re-asserted conditions shall include the retention floor as + well as status and age. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.5:** Each batch shall be atomic: after a + sweep is interrupted by shutdown or error, every batch that was applied is + complete, including its satellite rows, and no batch is partially applied. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.6:** Once a table holds no eligible row, + a further sweep against unchanged settings shall delete nothing from that table + and report zero deletions for it. This criterion is scoped to that drained + state and does not contradict AC-OFFICE-RUN-HISTORY-RETENTION-002.3: while a + table still reports remaining backlog, the next sweep is required to delete its + next batch, so "deletes nothing further" is not claimed of a backlogged table. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.7:** When a table's sweep fails with a + database error, the system shall record the failure for that table, continue + with the remaining in-scope tables, and retry the failed table on the next + scheduled sweep. A retention failure shall not fail backend startup, stop the + Office scheduler, or abort the remainder of the sweep. A batch abandoned + because its deletion could not be applied consistently shall be recorded as a + failure for that table rather than as backlog. Every table whose rows that + batch's transaction addressed shall report zero rows deleted for that sweep, so + no table reports rows that the rollback restored. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.8:** When retention is disabled, the + system shall run no sweep and delete no row, and shall still report retained + counts and threshold warnings. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.9:** When every in-scope table is empty + or holds no eligible row, the sweep shall complete reporting zero deletions, + without error and without a warning. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.10:** The first sweep after the backend + starts shall run after a short fixed delay rather than after a full sweep + interval, so that an install restarted more often than the interval still runs + retention. Later sweeps use the configured interval. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.11:** A sweep shall compute one cutoff + instant at its start and evaluate every table against that instant, so two + tables in one sweep cannot disagree about what "older than the window" means. + +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.12:** On PostgreSQL, where several backend + processes can share one database, sweep exclusivity shall hold across + processes and not only within one: a backend that cannot acquire the retention + lock shall skip its due sweep exactly as it would for a sweep already running + in its own process. On SQLite one backend process owns the database file, so + the in-process guard is sufficient. +- **AC-OFFICE-RUN-HISTORY-RETENTION-002.13:** Sweeps shall be scheduled fixed-delay: + the next sweep is armed when the previous one finishes, so a sweep running + longer than the interval delays its successor rather than causing an immediate + second one. A settings change re-arms the delay from the moment of the change. + Enabling retention that was disabled arms the next sweep at the same short + delay as AC-OFFICE-RUN-HISTORY-RETENTION-002.10 rather than at a full interval. + Each backend shall periodically reread the shared settings record, including + while retention is disabled, so a backend that did not serve a settings write + still adopts enablement and interval changes. + +### REQ-OFFICE-RUN-HISTORY-RETENTION-005: Integrity and database engine parity + +**Intent:** Retention leaves the database consistent and behaves identically on +both supported engines, including where the schema differs between them. + +#### Acceptance criteria + +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.1:** The system shall produce the same + observable retention outcome on SQLite and on PostgreSQL for the same settings + and the same starting rows: the same rows deleted, the same rows retained, the + same reported counts, and the same warnings. This shall hold on the backlog + path as well, because batch selection order is fixed by named columns in + AC-OFFICE-RUN-HISTORY-RETENTION-002.3 rather than left to the engine. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.2:** The system shall delete satellite + rows explicitly and shall not depend on a foreign-key cascade to remove them. + Neither engine declares a foreign key from a satellite table to `runs`. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.3:** After any sequence of sweeps, the + concurrency gate that finds a routine's active run by dispatch fingerprint + shall return exactly what it would have returned had no sweep run. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.4:** After any sequence of sweeps, the + lookup that closes out a routine run when its linked task reaches a terminal + step shall not resolve a deleted row to a different routine's run. A deleted + row resolves to nothing. +- **AC-OFFICE-RUN-HISTORY-RETENTION-005.5:** Retention shall not change the + behavior of deleting a routine or deleting a workspace. Those paths continue + to remove every row they remove today, including rows retention has not yet + reached. + +## Out of scope + +Each exclusion below is a decision, not an oversight. + +- **Automation run history (`automation_runs` and its tables).** Owned by the + automation system, and its published contract is that history is removed only + by an explicit per-run or delete-all action. It has the same unbounded-growth + gap and needs its own requirement; changing it here would silently break a + documented promise. +- **`office_activity_log`, inbox dismissals, and approval rows.** The + [inbox requirement](inbox.md) already states these accumulate indefinitely and + excludes activity-log retention. Reopening it belongs to that contract. +- **`office_cost_events`.** Deleting cost rows changes reported spend and budget + enforcement: a decision about financial records, not about storage. +- **Detecting a stuck routine.** The 323-row reference case came from a routine + that fired correctly and did nothing useful for 28 days. Bounding its history + does not detect it; a detector for that state is a scheduler or stall + visibility concern. +- **Reclaiming file bytes after deletion.** Deleting rows does not shrink a + SQLite file. `VACUUM` is already an operator action on the System pages and a + sweep does not trigger it. +- **Archival or export before deletion.** Deleted history is gone. An install + needing it kept sets a longer window or takes a backup. +- **Per-workspace or per-routine retention overrides.** Settings are + instance-wide here. The retention floor is already per owner, which covers what + a per-routine override would most often be used for. +- **A persisted history of retention sweeps.** The last sweep's result is held + in memory and reported alongside the durable warnings. A growing table + recording the work of the job that stops tables growing is the same defect in + a new place. +- **Tables outside Office.** Nothing here changes task, session, workflow, + plugin, or auth storage. + +## Prior art + +Receipts for both legs. The design document carries what each finding changed. + +**Our own wiki: not consulted, tool unavailable.** The `@henry` pin resolved +`~/.obsidian-wiki/config` to `config.henry`, giving +`OBSIDIAN_VAULT_PATH=/Users/henry/Documents/henry/wiki` and +`QMD_WIKI_COLLECTION=wiki`. Neither retrieval path ran: `qmd` and +`obsidian-wiki` are absent from `PATH`, no QMD MCP tool is registered in this +session, and the vault directory returns `Operation not permitted` both +sandboxed and unsandboxed, a macOS file-access restriction on this process +rather than a missing vault. The grep fallback is blocked the same way, so this +leg is a tooling gap, not evidence the wiki is silent on retention. + +**What other products shipped: consulted.** Queried the `saas-kb` server +(`search_fsm_docs`, `category: "ai_sdlc"`) three times: run-history retention and +database growth; session history retention and automatic deletion; and +scheduled-automation run-history limits. Relevance was low across all three, +itself a finding about corpus coverage. Two useful hits: **GitLab Duo** deletes +sessions 30 days after last activity, which anchors the default window here; and +**the Claude apps gateway** documents four tables with per-table windows +enforced by one hourly sweep, marking one table "until deleted via the API" +instead of giving it a window, which is the split this document draws between +history rows and live-state rows. + +Neither previewed before a policy's first deletion, nor warned ahead of the +window. Those are where this capability goes further, because an upgrade that +silently deletes a year of history on its first sweep is the failure mode a +shipped default carries. diff --git a/docs/specs/office/system-design/run-history-retention-operations.md b/docs/specs/office/system-design/run-history-retention-operations.md new file mode 100644 index 00000000000..f4195860f7c --- /dev/null +++ b/docs/specs/office/system-design/run-history-retention-operations.md @@ -0,0 +1,379 @@ +--- +status: current +system: office +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-003 + - REQ-OFFICE-RUN-HISTORY-RETENTION-004 +--- + +# Office Run History Retention Operations System Design + +## Purpose and boundaries + +This design covers the operator half of run history retention: the settings +record, the per-table preview marker, the reporting value, the health check, and +the read/write System page surface. The sweep itself, the eligibility +predicates, the batching, and the engine-parity guarantees are designed in [run +history retention](run-history-retention.md). + +The two documents share one component, `internal/office/retention`, and one +scheduler goroutine. They are split because they are two contracts with two +consumers: a scheduler deleting rows, and an operator reading a page. The split +also keeps each document inside the specification size limit without cutting +either contract. + +Adjacent contracts read and constrained but not owned: + +- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health` + response the System page's health card renders. +- `internal/system/settings.Store` — the key/value settings table, already used + by storage maintenance under one JSON key with normalization on read. + +## Component: the operator surface + +### Settings + +One `settings` key, `office_run_retention`, holding a JSON document, +read through a `Get`/`Save` pair with `Normalize` on both, matching +`internal/system/storage/settings.go`. Unparseable content returns the defaults +plus a sentinel error the caller turns into a health issue, never a boot failure +(AC-OFFICE-RUN-HISTORY-RETENTION-004.4). + +```json +{ + "enabled": true, + "sweep_interval_hours": 6, + "batch_limit": 5000, + "routine_runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 }, + "runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 }, + "run_events": { "warn_rows": 250000 } +} +``` + +`run_events` carries only a threshold: its lifetime is its run's, so it has no +window and no floor of its own. Ranges and rejection behavior are +AC-OFFICE-RUN-HISTORY-RETENTION-004.2 and .3; validation returns a field-named +error and writes nothing, as `validateRange` does for storage maintenance. +Writes are last-writer-wins through `Store.Save`; `CompareAndSwap` is not used +because these are operator-scale settings edited from one page, and an identical +repeated write is indistinguishable from no write +(AC-OFFICE-RUN-HISTORY-RETENTION-004.5). + +**Each sweep re-reads the settings from the store at its start, and skips if that +read fails.** The buffered +wake channel that re-arms the interval is process-local, so on a deployment where +several backends share one PostgreSQL database — the deployment +AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists for — it only ever reaches the +process that served the `PUT`. A backend relying on a cached effective value +could win the sweep lock still holding the policy the operator has just replaced +and delete rows the new policy retains. Re-reading costs one indexed key lookup +per sweep interval, against the risk of deleting history under a window the +operator already widened. The wake channel keeps its job — re-arming the timer +promptly in the process that saw the change — and stops being the only path by +which a change reaches a sweep. + +A failed or unparseable re-read **skips the sweep** rather than falling back to +the documented defaults. Everywhere else in this design an unreadable settings +document yields the defaults and a health issue +(AC-OFFICE-RUN-HISTORY-RETENTION-004.4), which is right for reporting and for +startup because neither deletes anything. It is wrong here: the default window is +30 days, an operator who widened theirs to 3,650 has by definition configured +something longer, and falling back would delete the decade of history they +configured the system to keep. Reading settings to *show* them fails open; +reading them to *delete by* fails closed. + + +### The preview marker + +A second `settings` key, `office_run_retention_preview_completed`, +holding a JSON object keyed by swept table name whose values are the timestamp +at which that table's preview completed: +`{"office_routine_runs": "...", "runs": "..."}`. + +The marker is **per table, not per database**. A single global flag is unsafe +against AC-OFFICE-RUN-HISTORY-RETENTION-002.7, which lets one table fail while +the sweep as a whole still completes: if `runs` errored during the preview sweep +and `office_routine_runs` succeeded, a global flag would be written anyway and +`runs` would delete for real on the next sweep having never shown an operator a +would-delete count — the exact failure this capability exists to prevent. With a +per-table marker, `runs` is simply previewed again next sweep +(AC-OFFICE-RUN-HISTORY-RETENTION-003.4). + +A table's entry is written only when that table's preview evaluation completed +successfully, is never cleared by a settings change or a restart, and is not +written by a deleting sweep. A table whose preview finds nothing still gets its +entry, so its next sweep deletes normally. + +If the key is present but unparseable, every swept table is treated as **not yet +previewed** and `office_retention_preview_unreadable` is raised +(AC-OFFICE-RUN-HISTORY-RETENTION-003.10). The asymmetry is deliberate: a +spurious re-preview deletes nothing and costs one sweep, whereas assuming a +preview had completed permits a first deletion no operator ever saw. The safe +direction is the one that cannot delete. + +The preview's per-table counts are uncapped by the batch limit +(AC-OFFICE-RUN-HISTORY-RETENTION-003.9): capping them would report 5,000 to an +operator holding 200,000 eligible rows, which is exactly the number the warning +exists to convey. + + +### Reporting + +An in-memory `LastSweep` value replaced wholesale at the end of each sweep that +actually ran: start and finish times, and per **reported** table the deleted +count, a backlog flag, and an error string. Deleted counts are counted from +**committed** transactions only: when a `runs` batch is abandoned and rolled back +(AC-OFFICE-RUN-HISTORY-RETENTION-002.7), the three satellite tables whose rows +that transaction addressed report **zero** deleted for that sweep, not the counts +their statements returned before the rollback. The failure is recorded against +`runs`, but the satellites must not show rows the rollback restored — the one +surface an operator has for "what actually happened" would otherwise be wrong +precisely in the failure case it exists for. Per **swept** table it also carries +whether that table was previewed in this sweep and, when it was, its +`WouldDelete` count — without that field the preview's headline numbers, which +AC-OFFICE-RUN-HISTORY-RETENTION-003.2 and +AC-OFFICE-RUN-HISTORY-RETENTION-003.9 require an operator to see, would have +nowhere to be read from. Because the preview marker is per table, `preview` is a +per-table flag rather than one flag for the sweep. Skips are held separately and +never overwrite this value. Nothing is persisted +(AC-OFFICE-RUN-HISTORY-RETENTION-004.6, and the "no persisted sweep history" +exclusion: a growing table recording the work of the job that stops tables +growing is the same defect in a new place). Before the first sweep the value is +absent, and the surface says so rather than rendering zeros +(AC-OFFICE-RUN-HISTORY-RETENTION-004.7). + +**Retained counts do not come from `LastSweep`.** They are a property of the +table, not of a sweep, and three ACs need them when no sweep has run at all: +AC-OFFICE-RUN-HISTORY-RETENTION-003.7 (threshold warning while retention is +disabled), AC-OFFICE-RUN-HISTORY-RETENTION-004.8 (surface readable while +disabled), and AC-OFFICE-RUN-HISTORY-RETENTION-002.8 (disabled means no sweep, +yet counts are still reported). A separate `RetainedCounts` value therefore +holds one count per thresholded table — produced by the census described next — +plus, for `office_routine_runs`, the top routine by retained rows for +AC-OFFICE-RUN-HISTORY-RETENTION-003.5's attribution. + +**The count evaluation is a status census, not a bare `COUNT(*)`.** For each +swept table it issues one +`SELECT status, COUNT(*) FROM
GROUP BY status`. The retained count is the +sum of those rows, so the census costs one scan rather than two, and the same +result set is what detects a status in neither the history nor the live-state set +and raises `office_retention_unknown_status:
` +(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). That detector has to live here rather +than in the sweep: the sweep's predicate selects `status IN ()` +and therefore structurally cannot observe a status it does not select. +`run_events` is thresholded but not swept and has no status column, so it keeps a +plain `COUNT(*)`. + +It is refreshed on the sweep interval by the same scheduler goroutine — on that +schedule whether or not retention is enabled, since a disabled install is +precisely the one whose tables grow unattended, and is also the only install +where an unrecognized status would otherwise never be noticed — and served from +memory in between. + +**The first evaluation runs at `Start`, not at the first sweep.** The first sweep +is deliberately delayed five minutes (AC-OFFICE-RUN-HISTORY-RETENTION-002.10); +hanging the first count off that tick would leave every restart with five minutes +of absent counts, and a fresh install with none at all until it had swept once — +while AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and -004.8 require the threshold +warning and the surface to work on exactly those installs. + +It runs **on the scheduler goroutine, not on the caller of `Start`**. The census +is three unbounded scans of the largest tables in the database, and the install +that most needs them is the one where they take longest; blocking `Start` on them +would make bounding these tables a cause of slow boots. `Start` returns +immediately, the tables read *not yet computed* until the first census lands +seconds later, and that state is already required and already renders honestly. + +`RetainedCounts` is therefore **tri-state per thresholded table**, not a number +that defaults to zero: *not yet computed*, *fresh as of T*, or *stale as of T*. +Zero is a real measurement and must not be how "nobody has counted yet" renders — +the same argument AC-OFFICE-RUN-HISTORY-RETENTION-004.7 makes for `LastSweep`, +and the reason AC-OFFICE-RUN-HISTORY-RETENTION-003.11 states it. The three states +are tracked **per table**, so one table's failing query neither discards nor +staleness-marks another's successful one; a failure with no prior success leaves +that table at *not yet computed* rather than fabricating a zero. + +The evaluation is explicitly **not** computed per health poll or per page load: +`/api/v1/system/health` is polled by every open browser tab, and issuing three +unbounded `COUNT(*)`s against the largest tables in the database on each poll +would make this feature a cause of the load it exists to prevent +(AC-OFFICE-RUN-HISTORY-RETENTION-003.11). + +Three surfaces, in descending durability: + +1. **Structured logs.** One `info` line per sweep. One `warn` line per condition + in REQ-OFFICE-RUN-HISTORY-RETENTION-003, carrying the table, the counts, and + the threshold. Always on, in every profile. +2. **Health issues.** The package implements `health.Checker` with + `Name() = "Office run retention"` and `Category() = "office"`, returning a + `health.Issue` per active condition with + `FixURL = "/settings/system/data-storage"`, which is the live route + registered in `apps/web/src/settings-routes.tsx` — note the suffix, as + `/settings/system/data` is not a registered path and `/settings/system/database` + is only a redirect to it. `internal/health/checks_test.go` pins fix URLs to + live routes ("expectedGitHubFixURL is the live route in ..."), and this one + is pinned the same way. This is the production-visible surface + required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 and is why the debug + metrics endpoint is not it: `/debug/vars` is gated on the dev profile. + Issue ids are stable and one per condition: + `office_retention_preview_pending`, `office_retention_backlog:
`, + `office_retention_threshold:
`, `office_retention_disabled:
`, + `office_retention_settings_invalid`, `office_retention_failed:
`, + `office_retention_preview_unreadable`, and + `office_retention_unknown_status:
`. + `Check` returns issues sorted by id, matching `storage.Runtime.Check`, so the + health card's order is stable across polls. + For `office_routine_runs`, the threshold issue's message names the routine + holding the largest share of retained rows and that share + (AC-OFFICE-RUN-HISTORY-RETENTION-003.5) — attribution computed from the same + retained-count query, not a second detector. +3. **expvar.** Counters under `office_retention_*` following + `office/scheduler/metrics_vars.go`. Development convenience only; nothing in + REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on it. + + +### HTTP and frontend + +`GET /api/v1/system/retention` returns effective settings, `LastSweep`, the +separately-held skip record, and `RetainedCounts`. + +`PUT /api/v1/system/retention` **replaces the whole document**; it is not a +merge patch. A field the caller omits takes its documented default rather than +its stored value, so the same request body always yields the same stored +document and a repeated identical write is genuinely a no-op — which is what +AC-OFFICE-RUN-HISTORY-RETENTION-004.5's "identical repeated write changes +nothing and returns the same normalized document" requires. A merge would break +that: under a merge, whether a request is a no-op depends on what was stored +before it. An unrecognized field, a numeric field whose value is not a whole +number, a field present with a JSON `null`, and a field whose value is of the +wrong type, are each rejected with a field-named 400 and nothing is written +(AC-OFFICE-RUN-HISTORY-RETENTION-004.9); rejecting rather than ignoring an +unknown field means a client that misspells `window_days` is told so instead of +silently getting the default. `null` needs saying because this endpoint is a full +replace: since an omitted field deliberately means "take the default", the +tempting reading of `null` is the same one, and that reading turns +`{"runs": {"window_days": null}}` into a silent 3,650-to-30-day reduction that +destroys a decade of history on the next sweep. Decode into pointer fields and +reject an explicit null, rather than into value fields where `null` and omission +are indistinguishable. Successful writes return the normalized document. + +Both routes are admin-scoped like the other System routes and are readable while +retention is disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8). + +One card on **Settings > System > Data & Logs**, the page served at +`/settings/system/data-storage` and rendered by +`apps/web/components/settings/system/data-logs-settings.tsx`, beside +`database-stats-card.tsx`: the enable toggle, the numeric fields, and the +last-sweep readout. It follows the storage-maintenance cards' shape. All new +copy goes through `t()` and must ship in `pt-pt`, `zh-cn`, `zh-hk`, and `zh-tw`; +`pnpm run i18n:check` and the new-code ratchet gate the build. Health issue +titles and messages stay English, matching every other backend-produced +`health.Issue`. + + +## Ordering, concurrency, and failure + +| Question | Answer | AC | +|---|---|---| +| Concurrent settings writes | Last-writer-wins; an identical repeat changes nothing | 004.5 | +| Settings unreadable | Defaults used, health issue raised, sweep proceeds | 004.4 | +| Settings out of range | Rejected with the field named, stored settings unchanged | 004.3 | +| Retention disabled | No sweep, no deletion; counts and threshold warnings still reported | 002.8, 003.7, 004.8 | +| Preview scope | Per swept table, not per database; a table that failed its preview is previewed again | 003.4 | +| Preview marker unreadable | Treated as not previewed; re-preview deletes nothing | 003.10 | +| Retained counts | Table property from a `GROUP BY status` census; first evaluation at `Start`, then on the sweep interval; served from memory; computed even while disabled | 003.11 | +| Counts before the first evaluation, or first evaluation fails | Reported *not yet computed* per table, never as zero; state tracked per thresholded table | 003.11, 004.7 | +| Unknown status detected | By the census, not the sweep predicate; one issue per table listing every unrecognized status in ascending order | 001.10 | +| Preview on a table with more eligible rows than the batch limit | Reports the full eligible count; never reported as backlog | 003.6, 003.9 | +| Settings changed on another backend | Each sweep re-reads settings at its start, so a process that did not serve the write still uses the new policy | 004.5 | +| Settings unreadable at sweep start | Sweep skipped and recorded as skipped; never run under the shorter default window | 004.5, 004.4 | +| Batch abandoned and rolled back | Satellite tables report zero deleted, not their pre-rollback statement counts | 002.7, 004.6 | +| Settings write shape | Full replace; omitted field takes its default; unknown field 400 | 004.9 | + +## Testing + +Unit tests in `internal/office/retention`, plus the frontend checks below. + +- The first sweep on a seeded database deletes nothing and reports a + would-delete count; the second deletes (003.1, 003.4). +- A preview on a database with more eligible rows than the batch limit reports + the full eligible count (003.9). +- A table that errors during the sweep in which it was being previewed is + previewed again on the next sweep instead of deleting, while a sibling table + that succeeded proceeds to delete (003.4). +- A preview marker that is present but unparseable causes a re-preview, not a + deletion, and raises `office_retention_preview_unreadable` (003.10). +- Retained counts and the threshold warning are produced on a fresh install + before any sweep, and while retention is disabled (003.7, 003.11, 004.8), and + a health poll does not issue a table count. +- Counts are available immediately after `Start`, without advancing any clock to + the first sweep's delay (003.11). A test that waits out the delay would pass + against an implementation that hangs the first count off the sweep tick, which + is the defect this asserts against. +- Before the first evaluation, and when the first evaluation fails with no + earlier success, the surface reports the table as *not yet computed* and not as + zero (003.11, 004.7). +- With one thresholded table's count query failing and the other two succeeding, + the two keep fresh counts and only the failing one is marked, and the threshold + warnings derived from the successful counts still fire (003.11). +- A swept table holding a status in neither status set raises + `office_retention_unknown_status:
` from the census while retention is + **disabled** and no sweep has ever run (001.10, 003.11). +- A preview on a table with more eligible rows than the batch limit reports the + full eligible count and raises no backlog warning (003.6, 003.9). +- A settings change written through one store handle is used by a sweep driven + from a second handle that never saw the change notification (004.5). +- A backend census refresh adopts a settings change written by another backend, + including when retention was disabled, and arms the appropriate sweep timer + (002.13, 004.5). +- A sweep whose settings read fails is skipped and recorded as skipped, and + deletes nothing under the default window (004.5). +- A `runs` batch abandoned after its retry reports zero deleted for the three + satellite tables rather than their pre-rollback counts (002.7, 004.6). +- A recorded skip leaves the previous `LastSweep` readable and unchanged (002.2, + 004.7). +- A settings write omitting a field stores that field's default; a write with an + unknown field or a fractional number is rejected 400 naming the field and + stores nothing; the same write applied twice is a no-op (004.9). + +Commands: + +``` +cd apps/backend +go test ./internal/office/... -race -count=1 +KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... -count=1 +cd apps/web && pnpm run typecheck && pnpm run i18n:check +``` + +## Rejected alternatives + +- **One global "preview completed" flag.** Simpler, and unsafe: because + AC-OFFICE-RUN-HISTORY-RETENTION-002.7 lets one table fail while the sweep + completes, a global flag is written even when a table never got its preview, + and that table then deletes for real having shown the operator nothing. The + per-table marker costs one JSON object. +- **Compute retained counts inside the health check.** Direct, and it puts three + unbounded `COUNT(*)`s on a route every open browser tab polls. Refreshing on + the sweep interval and serving from memory gives the same number without + making the bound-the-tables feature a source of load on those tables. +- **A merge-patch settings write.** Under a merge, whether a request is a no-op + depends on what was stored before it, which + AC-OFFICE-RUN-HISTORY-RETENTION-004.5 forbids. +- **Deletion off by default.** Safe, and it means the gap stays open on every + install that never visits the settings page. The per-table preview gives the + same protection without that outcome. +- **Report warnings only through `/debug/vars`.** That endpoint is gated on the + dev profile, so on a production build the warnings would not exist + (AC-OFFICE-RUN-HISTORY-RETENTION-003.8). + +## Prior art, applied + +`internal/system/storage` supplies the settings storage and normalization +pattern, the hours-based interval with min and max bounds, and the +`health.Checker` route to a production-visible warning; `storage.Runtime.Check` +also sorts its issues by id, which this check matches so the health card's order +is stable across polls. + +Neither GitLab Duo nor the Claude apps gateway previews before a policy's first +deletion or warns ahead of the window. Those are this capability's additions, +and they are the whole reason this document is separate from the sweep's. diff --git a/docs/specs/office/system-design/run-history-retention.md b/docs/specs/office/system-design/run-history-retention.md new file mode 100644 index 00000000000..c0831c0ad1d --- /dev/null +++ b/docs/specs/office/system-design/run-history-retention.md @@ -0,0 +1,571 @@ +--- +status: current +system: office +requirements: + - REQ-OFFICE-RUN-HISTORY-RETENTION-001 + - REQ-OFFICE-RUN-HISTORY-RETENTION-002 + - REQ-OFFICE-RUN-HISTORY-RETENTION-005 +--- + +# Office Run History Retention System Design + +## Purpose and boundaries + +This design adds one scheduled sweep that deletes aged Office run history from +five tables. It changes no run lifecycle, no routine dispatch decision, and no +task, session, or checkout. + +Office owns the sweep because the eligibility rules are defined by Office +primitives. The settings record, the health check, the reporting value, and the +System page surface are the operator half of the same capability and are +designed in [run history retention +operations](run-history-retention-operations.md), which is where every +requirement in REQ-OFFICE-RUN-HISTORY-RETENTION-003 and -004 is satisfied. + +Adjacent contracts read and constrained but not owned: + +- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health` + response the System page's health card renders. +- `internal/system/settings.Store` — the key/value settings table, already used + by storage maintenance under one JSON key with normalization on read. +- `internal/runs/repository/sqlite` — the `runs` queue and `run_events` access + methods. +- `internal/office/repository/sqlite` — the schema owner for `runs`, + `run_events`, `office_run_route_attempts`, `office_run_skills`, and + `office_routine_runs`. + +## Measured starting state + +Read from the reference install's SQLite database on 2026-09-09. These numbers +set the defaults and are the baseline any regression test can be written +against. + +| Table | Rows | Notes | +|---|---|---| +| `office_routine_runs` | 326 | 323 `coalesced`, 3 `task_created` | +| `runs` | 53 | | +| `run_events` | 340 | across 53 runs, mean 6.4 per run | +| `office_run_route_attempts` | 55 | | +| `office_run_skills` | 382 | | + +The 323 `coalesced` rows span 2026-08-03 to 2026-08-31 and belong to a single +routine that is now `paused`. Orphan `run_events` today: zero, because nothing +has ever deleted a `runs` row. + +## Schema facts this design depends on + +Verified by reading the schema owners, not assumed. + +- `office_routine_runs` (`office/repository/sqlite/base.go`) has + `FOREIGN KEY (routine_id) REFERENCES office_routines(id) ON DELETE CASCADE` + on both engines, and SQLite opens with `_foreign_keys=on` + (`internal/db/sqlite.go`). Its two indexes are partial and serve the dispatch + gate, not an age scan: `idx_office_routine_runs_active_fingerprint` + (`WHERE status = 'task_created'`) and `idx_office_routine_runs_linked_task` + (`WHERE linked_task_id != ''`). +- `run_events`, `office_run_route_attempts`, and `office_run_skills` declare + **no foreign key to `runs`** on either engine. Confirmed in + `office/repository/sqlite/base.go` and in the PostgreSQL conformance snapshot + `internal/persistence/storeconformance/testdata/upgrades/v0.93.0/postgres.sql`, + where `run_events` has only a primary key and one index. A cascade would + therefore delete nothing; satellite deletion must be explicit + (AC-OFFICE-RUN-HISTORY-RETENTION-005.2). +- `run_events.seq` is assigned by `AppendRunEvent` as + `COALESCE(MAX(seq) + 1, 0)` scoped to the run. Deleting a live run's whole + timeline restarts its sequence at zero, and the run detail view's incremental + tail reads `WHERE seq > afterSeq`. This is why + AC-OFFICE-RUN-HISTORY-RETENTION-001.6 forbids event deletion outside the + transaction that deletes the run. +- `runs` indexes are `idx_run_status_requested (status, requested_at)` and the + partial unique `idx_run_idempotency`. Nothing indexes `finished_at`. +- `office_agent_pause_recoveries.failed_run_id` identifies failed runs that an + active pause recovery still needs. It has no foreign key, so retention uses a + correlated `NOT EXISTS` predicate and the supporting + `idx_office_agent_pause_recoveries_failed_run` index. +- `ScheduleRetry` (`runs/repository/sqlite/runs.go`) sets + `status = 'queued', finished_at = NULL` on an existing run, keyed by id with + **no status guard in its `WHERE` clause**. Any terminal run can therefore + become live again at any moment, which is the race + AC-OFFICE-RUN-HISTORY-RETENTION-002.4 closes. Because the guard is absent, a + `cancelled` row is resurrectible on exactly the same terms as a `failed` one, + so admitting `cancelled` to the history set adds no new race — it is covered + by the same re-assertion. +- `CancelRunsWhere` (`runs/repository/sqlite/cancel.go`) is documented as "the + single writer of the terminal cancel state on the runs table". It sets + `status = 'cancelled', cancel_reason = ?, finished_at = ?` and is guarded by + `status IN ('queued', 'claimed')`, so a cancelled row is always terminal and + always carries a completion timestamp. It is production-reachable through + `CancelRunsForTasks` from `office/service/tree_controls.go` and + `office/repository/sqlite/participants.go`. It writes the literal string, so + `office/models/enums.go`'s four-value `RunStatus` block does not enumerate it: + the database has five `runs` statuses, not four. This is why + AC-OFFICE-RUN-HISTORY-RETENTION-001.2 classifies `cancelled` as history and + AC-OFFICE-RUN-HISTORY-RETENTION-001.10 makes any sixth value fail safe. +- Every production writer of a terminal `runs` status stamps the completion + timestamp: `FinishRun` sets `finished_at = now`, `MarkRunFailed` sets + `finished_at = COALESCE(finished_at, now)`, and `CancelRunsWhere` sets it + outright. Only the test helper `SetRunStatusForTest` can leave a terminal row + with a null `finished_at`. The `COALESCE(finished_at, requested_at)` fallback + in AC-OFFICE-RUN-HISTORY-RETENTION-001.3 is therefore defensive rather than a + routine path — but it is also what makes the ordering key non-null, which is + what keeps the floor deterministic across engines (see below). +- `runs.requested_at` and `office_routine_runs.created_at` are both + `TIMESTAMP NOT NULL`, so the completion-time expression is total. This matters + for parity, not just tidiness: SQLite and PostgreSQL differ in where they sort + nulls by default, so an `ORDER BY` over a nullable timestamp would rank the + floor differently on the two engines from identical data. +- `CleanExpired` (`runs/repository/sqlite/runs.go`) already deletes terminal + `runs` rows older than a cutoff and **has no production caller** — only + tests reach it. It deletes only `runs`, so wiring it as-is would orphan every + satellite row. It is superseded by the batched, satellite-aware delete below + rather than reused. + +## Component: `internal/office/retention` + +One package holding policy, settings, the sweep, and the health check. + +### Ownership and lifecycle + +A single goroutine owner modelled on `internal/system/storage.Scheduler`: a +`Start(ctx)` that is a no-op when already running, a `Stop()` that cancels and +joins, and a buffered wake channel so a settings change re-arms the interval +without interrupting a sweep in progress +(AC-OFFICE-RUN-HISTORY-RETENTION-004.5). `Stop` is joined from the same place +that stops the Office scheduler. + +The loop is deliberately not the Office run-processing tick +(`office/service/scheduler_integration.go`, `DefaultTickInterval = 5s`). That +tick already carries `RecoverStale` and `ReapStaleCheckouts` unthrottled and is +the run-claim hot path; on SQLite a bulk delete there contends with the single +writer that claims runs, 17,280 times a day, for an input that changes on a +scale of days (AC-OFFICE-RUN-HISTORY-RETENTION-002.1). + +A `sweeping bool` guarded by the same mutex makes a due sweep a skip rather than +a second goroutine (AC-OFFICE-RUN-HISTORY-RETENTION-002.2). That guard is +process-local, which is sufficient on SQLite, where the database file is owned +by one backend process. On PostgreSQL, where several backends can share one +database, it is not: two backends would each see `sweeping == false` and sweep +the same rows concurrently. + +The sweep therefore also takes a PostgreSQL advisory lock, and it must be a +**session-scoped, non-blocking** one, which is a different shape from every +existing advisory lock in this repository: + +``` +conn := db.Conn(ctx) // one dedicated connection +SELECT pg_try_advisory_lock(:retention_key) // boolean, returns immediately +... whole sweep, every table, every batch, on the pool ... +SELECT pg_advisory_unlock(:retention_key) // on that same connection +conn.Close() // deferred +``` + +Both properties are load-bearing and neither is optional: + +- **Session-scoped, not transaction-scoped.** The sweep is multi-transaction by + construction: AC-OFFICE-RUN-HISTORY-RETENTION-002.5 makes each batch its own + transaction, and AC-OFFICE-RUN-HISTORY-RETENTION-002.7 requires one table's + failure to leave another table's committed deletions intact, which forbids + wrapping the sweep in a single transaction. A `pg_advisory_xact_lock` is + released when its transaction ends, so it would protect one batch and then let + a second backend in between tables — the interleaving + AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent. The lock is held on a + connection checked out for the sweep and released in a `defer`; the batches + themselves continue to use the pool. +- **`try`, not the blocking form.** AC-OFFICE-RUN-HISTORY-RETENTION-002.12 says a + backend that cannot acquire the lock *skips*. `pg_advisory_xact_lock` and + `pg_advisory_lock` wait instead of failing, which would convert a concurrent + sweep into a queued one and eventually stall the scheduler behind a long sweep. + `pg_try_advisory_lock` returns `false` immediately; on `false` the backend + records a skip and returns, exactly as for a local concurrent sweep. + +This deliberately departs from the established repo pattern +`SELECT pg_advisory_xact_lock(hashtextextended(?, 0))` +(`office/repository/sqlite/participants.go`, `internal/secrets/sqlite_store.go`, +`internal/workflow/repository/phase2_sqlite.go`), and the departure is the point: +each of those call sites performs its entire unit of work inside the one +transaction that holds the lock, and each *wants* to wait rather than skip. The +sweep can do neither. The key is derived the same way, `hashtextextended` over a +constant distinct from every key those sites use, so retention never contends +with participant-seat, secret-transfer, or workflow-phase locking. + +**If the lock connection drops mid-sweep**, PostgreSQL releases the session's +advisory locks as part of ending the session, so a crashed or partitioned backend +cannot wedge retention permanently — that self-healing is the reason for a +session lock rather than a lease row in the `settings` table, which would need its +own expiry and its own stale-holder rule. The cost is that the surviving sweep no +longer holds exclusivity without knowing it, so the sweep **verifies the lock +connection is still alive between tables** and, if it is not, stops before the +next table, records that table as skipped rather than failed, and returns. It +does not attempt to re-acquire mid-sweep: a re-acquisition after another backend +has taken the lock would produce exactly the concurrent sweep this protects +against. Batches already committed stay committed, which +AC-OFFICE-RUN-HISTORY-RETENTION-002.5 permits. + +A skip is recorded as its own value — a skip counter and a last-skip timestamp — +and does **not** overwrite `LastSweep`. `LastSweep` holds the last sweep that +actually ran, so a burst of skips cannot blank the operator's view of the last +real result (AC-OFFICE-RUN-HISTORY-RETENTION-002.2, and +AC-OFFICE-RUN-HISTORY-RETENTION-004.7, which forbids rendering a +never-swept-looking surface when a sweep has in fact run). + +The first sweep after `Start` is armed at a fixed 5-minute delay rather than a +full interval (AC-OFFICE-RUN-HISTORY-RETENTION-002.10). Arming at the interval, +as the storage scheduler does with its 24-hour default, means an install +restarted more often than the interval never sweeps at all; 5 minutes keeps +startup clear of schema init and run recovery without depending on uptime. Each +sweep computes `time.Now().UTC()` once and passes that instant to every table, +so two tables in one sweep cannot disagree about the cutoff +(AC-OFFICE-RUN-HISTORY-RETENTION-002.11). + +Scheduling is **fixed-delay, not fixed-rate**: the next sweep is armed when the +previous one returns, so a sweep that overruns its interval delays its successor +instead of causing an immediate second one (which the concurrency guard would +only skip anyway, turning a slow sweep into a stream of skips). A settings +change re-arms the delay from the moment of the change rather than from the last +sweep, and enabling retention that was disabled arms at the same 5-minute delay +as a fresh start rather than at a full interval — otherwise an operator who +enables retention on a 168-hour interval waits a week to see whether it works +(AC-OFFICE-RUN-HISTORY-RETENTION-002.13). + +The census timer also re-reads shared settings. It runs while retention is +disabled, so other backends discover enablement and interval changes and re-arm +their timers. + +### Eligibility, expressed once + +Two predicates, each defined in exactly one place and reused by the count, the +preview, and the delete. + +**Routine runs.** History statuses are `skipped`, `coalesced`, `failed`, `done`, +`cancelled`. `received` and `task_created` are absent by construction, which is +what makes AC-OFFICE-RUN-HISTORY-RETENTION-005.3 hold: the dispatch gate +`GetActiveRunForFingerprint` reads only `status = 'task_created'`, so no sweep +can change its answer. + +``` +DELETE FROM office_routine_runs +WHERE id IN ( + SELECT id FROM ( + SELECT id, + ROW_NUMBER() OVER ( + PARTITION BY routine_id + ORDER BY COALESCE(completed_at, created_at) DESC, id DESC + ) AS rn + FROM office_routine_runs + WHERE status IN () + ) ranked + WHERE rn > :floor + AND completion_time < :cutoff + ORDER BY completion_time ASC, id ASC + LIMIT :batch +) +AND status IN () +AND COALESCE(completed_at, created_at) < :cutoff +``` + +where the ranked subquery also projects +`COALESCE(completed_at, created_at) AS completion_time`. + +Two orderings appear here and they are not the same ordering; conflating them is +the defect this section exists to prevent. + +- The **`ORDER BY` inside `ROW_NUMBER()`** ranks rows *within* a partition so the + floor keeps the newest per owner: `completion_time DESC, id DESC` + (AC-OFFICE-RUN-HISTORY-RETENTION-001.4). +- The **`ORDER BY` on the outer select** decides *which* eligible rows a + batch-limited sweep takes: `completion_time ASC, id ASC`, oldest first + (AC-OFFICE-RUN-HISTORY-RETENTION-002.3). The window function's ordering does + not reach the outer `LIMIT`, so without this clause the engine is free to + return any subset and the two engines may drain a backlog differently from + identical data — which would contradict + AC-OFFICE-RUN-HISTORY-RETENTION-005.1 and make the parity test below flake for + a reason unrelated to a genuine engine difference. + +`id` is a UUID and so is not chronological; it is used only as a total, stable +tiebreak for equal timestamps, in both orderings. + +The trailing `AND` clauses are the re-assertion required by +AC-OFFICE-RUN-HISTORY-RETENTION-002.4. For this table the whole ranked subquery +is re-evaluated inside the `DELETE`, so the floor is re-asserted atomically along +with status and age; the two-phase `runs` path below has to do that explicitly. + +Window functions are available on both engines: SQLite 3.54.0 through +`github.com/mattn/go-sqlite3 v1.14.33`, verified by running this exact +`ROW_NUMBER() OVER (PARTITION BY routine_id ...)` against the reference +database. + +**Runs.** History is `status IN ('finished','failed','cancelled')`, partitioned +by `agent_profile_id`, ranked `COALESCE(finished_at, requested_at) DESC, id DESC` +and batch-ordered `COALESCE(finished_at, requested_at) ASC, id ASC`. A failed +run named by an active `office_agent_pause_recoveries.failed_run_id` is +protected by a `NOT EXISTS` clause in this predicate. Count, selection, and +delete use the same clause, so a preview cannot promise deletion of a run that +the recovery flow still needs. There is no +`finished_at IS NOT NULL` conjunct: requiring one would make a terminal row with +an unset stamp immortal and unobservable, which +AC-OFFICE-RUN-HISTORY-RETENTION-001.3 forbids. `queued` and `claimed` are +absent, which covers a routing-parked run whose `earliest_retry_at` is far in +the future (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). + +A status in neither set is treated as live state and raises +`office_retention_unknown_status:
` +(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). Both sets are closed and asserted +against `office/models/enums.go` plus the literal-SQL writers in a test, so a +sixth status cannot enter the database without failing that test — the failure +mode that hid `cancelled` in the first place. + +**That warning needs a producer, and the eligibility predicate cannot be it.** +The predicate selects `status IN ()`, so a row holding an +unrecognized status is never selected, never counted, and never seen: a fail-safe +whose only detector is a CI test fires on the developer's machine and stays +silent on the install that actually has the row. The producer is instead the +**status census** — one + +``` +SELECT status, COUNT(*) FROM GROUP BY status +``` + +per swept table, run by the retained-count evaluation described in [run history +retention operations](run-history-retention-operations.md#reporting). Three +consequences follow from siting it there rather than in the sweep, and each one +closes a hole: + +- The census **subsumes the retained count** rather than adding a second scan: + the retained count is the sum of the census rows, so one `GROUP BY` yields + both. (`run_events` is thresholded but not swept and has no status column, so + it keeps a plain `COUNT(*)`.) +- It runs **whether or not retention is enabled**, because the count evaluation + does (AC-OFFICE-RUN-HISTORY-RETENTION-003.11). A disabled install is precisely + where an unrecognized status would otherwise never be noticed, since + AC-OFFICE-RUN-HISTORY-RETENTION-002.8 means no sweep runs there at all. +- It sees a status **because the row exists**, not because the row was + selectable, which is the property the eligibility predicate structurally + cannot have. + +One issue per table, not one per status: a table with several unrecognized +statuses raises the single id `office_retention_unknown_status:
` whose +message lists every unrecognized status **in ascending lexicographic order** with +its row count. Ordering is named because the message is compared across health +polls; an unordered list would make a stable condition look like a changing one. + +### Deleting a run + +Per batch, one transaction, satellites first, run last: + +1. Select up to `batch_limit` eligible run ids, excluding runs named by an + active `office_agent_pause_recoveries.failed_run_id`. +2. `DELETE FROM run_events WHERE run_id IN (...)` +3. `DELETE FROM office_run_route_attempts WHERE run_id IN (...)` +4. `DELETE FROM office_run_skills WHERE run_id IN (...)` +5. `DELETE FROM runs WHERE id IN (:selected_ids) AND id IN ()` — the re-assertion is the **whole ranked subquery**, not just the + status and cutoff conjuncts, so the retention floor is re-evaluated at delete + time along with them. Unlike the single-statement `office_routine_runs` + delete, this path selected its ids in a separate earlier statement, so a + `ScheduleRetry` in between can re-rank a partition and push a row that was + `rn > floor` at selection to `rn <= floor` now. Re-asserting only status and + age would delete a row that has since become floor-protected + (AC-OFFICE-RUN-HISTORY-RETENTION-002.4). + +Step 5 can delete fewer rows than steps 2 to 4 addressed, when a `ScheduleRetry` +resurrected a run between selection and delete. That is the correct outcome for +AC-OFFICE-RUN-HISTORY-RETENTION-002.4 only if the whole batch rolls back rather +than leaving a live run without its timeline. **The transaction is therefore +rolled back and retried once with a fresh selection when the step 5 row count +does not match the selected id count**; a second mismatch rolls back again and +abandons the batch. + +An abandoned batch is recorded as a **failure** for that table, raising +`office_retention_failed:
`, not as backlog +(AC-OFFICE-RUN-HISTORY-RETENTION-002.7). The distinction is not cosmetic: +backlog means work correctly deferred by the batch limit and is expected on a +large install, so routing this case there would file the one genuinely dangerous +outcome under the one routine one. The table is retried on the next scheduled +sweep either way, but only the failure classification tells an operator that a +deletion could not be applied consistently. + +Step 5 can only ever delete *fewer* rows than steps 2 to 4 addressed, never +more, because it is bounded by the same selected id set. This is the one place +where the naive implementation silently corrupts a live run, and it is the +reason satellite deletion is not a separate statement outside the transaction. + +`run_events` is never addressed by any predicate other than membership in this +id set (AC-OFFICE-RUN-HISTORY-RETENTION-001.6). There is no age-based delete on +`run_events`. + +### Indexes to add + +Retention adds indexes that serve the sweep. + +- `idx_office_routine_runs_retention ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)` +- `idx_runs_retention ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)` +- `idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)` + +Added through the existing `office/repository/sqlite` schema path so both the +fresh-install `CREATE` and the upgrade path get them, and recorded in the +conformance fixtures. + +## Ordering, concurrency, and failure + +| Question | Answer | AC | +|---|---|---| +| Sweep ordering across tables | `office_routine_runs`, then `runs` with its satellites. Independent sets; the order is fixed only so results and logs are reproducible. | 002.6 | +| Floor ordering | `COALESCE(completed_at, created_at) DESC, id DESC` / `COALESCE(finished_at, requested_at) DESC, id DESC` | 001.4 | +| Two sweeps due at once | Second is skipped, not queued, and recorded as skipped | 002.2 | +| Sweep vs. live writer | Selection is a snapshot; status, age **and floor** are re-asserted in the delete; a batch whose step 5 count disagrees is rolled back | 002.4, 002.5 | +| Sweep interrupted | Each batch is one transaction; applied batches are whole, the interrupted one is not applied | 002.5 | +| First sweep after startup | Armed at a fixed 5-minute delay, not a full interval | 002.10 | +| Cutoff instant | Computed once per sweep, shared by every table | 002.11 | +| Re-run once no eligible rows remain | Deletes nothing further, reports zero. While backlog remains, the next sweep deletes the next batch by 002.3 — the two are not in tension because 002.6 is scoped to the drained state | 002.6, 002.3 | +| One table errors | That table is recorded failed; remaining tables continue; retried next sweep; startup unaffected | 002.7 | +| Empty tables | Zero deletions, no error, no warning | 002.9 | +| Deleted coalesce target | Allowed; `coalesced_into_run_id` is provenance, read by nothing | 001.9 | +| Routine paused | Irrelevant to eligibility | 001.7 | +| Cancelled run | History, like `finished`/`failed`; its writer stamps the completion timestamp | 001.2 | +| Terminal row, null completion stamp | Still history; dated by `created_at` / `requested_at` | 001.3 | +| Status in neither set | Treated as live state, never deleted, warned | 001.10 | +| Which rows a batch-limited sweep takes | `completion_time ASC, id ASC` — oldest first, named columns | 002.3, 005.1 | +| Batch abandoned after retry | Recorded as that table's failure, not as backlog | 002.7 | +| Preview finds more eligible rows than the batch limit | Not backlog. A preview deletes nothing by design, so "retention is behind" would be false; it reports the full eligible count instead | 002.3, 003.6, 003.9 | +| Sweep skipped | Recorded separately; never overwrites the last real `LastSweep` | 002.2, 004.7 | +| Two backends, one PostgreSQL | Session-scoped `pg_try_advisory_lock` held on a dedicated connection for the whole sweep; the loser skips without waiting | 002.12 | +| Lock connection drops mid-sweep | PostgreSQL releases the lock with the session; the sweep stops before the next table and records it skipped, and does not re-acquire | 002.12 | +| Scheduling model | Fixed-delay from the end of the previous sweep; a settings change re-arms from the change | 002.13 | + +## Testing + +Unit and repository tests in `internal/office/retention` and +`internal/office/repository/sqlite`, plus the persistence gates. + +Behaviors that must have a test, because each is a way the naive implementation +is wrong: + +- A `task_created` routine run older than any window survives, and + `GetActiveRunForFingerprint` still finds it (001.1, 005.3). +- A `queued` run with `finished_at` cleared by `ScheduleRetry` survives (001.2). +- A run resurrected between selection and delete leaves both the run and its + full `run_events` timeline intact (002.4, and the rollback above). +- After a sweep, no `run_events`, `office_run_route_attempts`, or + `office_run_skills` row references a missing `runs` row (001.5, 005.2). +- No `run_events` row of a surviving run is ever deleted (001.6). +- A routine with 3 history rows all older than the window keeps all 3 under a + floor of 50; a routine with 200 keeps exactly 50 (001.4). + +- Two sweeps triggered concurrently produce one sweep and one recorded skip + (002.2). +- A sweep hitting the batch limit reports backlog and the next sweep continues, + and its satellite rows are deleted in full rather than capped (002.3, 003.6). + +- A terminal row whose completion timestamp is unset is dated by `created_at` + or `requested_at` and ages out rather than being retained forever (001.3). + This test is only satisfiable because AC-OFFICE-RUN-HISTORY-RETENTION-001.2 + classifies history by status alone; an implementation that also required + `finished_at IS NOT NULL` would retain the row forever and fail here. +- A `cancelled` run older than the window is deleted together with its satellite + rows, and a `cancelled` run inside the floor is retained (001.2). Seed it + through the real cancel path, not by writing the status directly, so the test + fails if that path stops stamping the completion timestamp. +- A `runs` row holding a status in neither the history set nor the live-state set + survives every sweep at any age and raises + `office_retention_unknown_status:runs` (001.10). Three companions: the same row + raises the same issue with retention **disabled**, where no sweep runs at all; + a table holding two unrecognized statuses raises **one** issue listing both in + ascending order with their counts; and a test asserts the two status sets + together cover every value in `office/models/enums.go` *and* every status + literal written by SQL in `internal/runs/repository/sqlite`, so a future sixth + value cannot be added silently — the exact gap through which `cancelled` was + missed. +- With more eligible rows than the batch limit, the sweep deletes the oldest + eligible rows first, and the same seed data yields the identical deleted set on + SQLite and PostgreSQL (002.3, 005.1). Without the outer `ORDER BY` this test is + the one that fails. +- A row that becomes floor-protected between selection and delete is not deleted + by the `runs` path (002.4). + +- A batch abandoned after its retry is reported as that table's failure and not + as backlog (002.7). + +- Two backends against one PostgreSQL run one sweep, and the loser records a + skip (002.12). The seed must give the winner **more than one table** to sweep, + so that a transaction-scoped lock — which would release between tables and let + the loser in — fails this test rather than passing it. A companion test drops + the winner's lock connection mid-sweep and asserts it stops before the next + table and does not re-acquire. +- A `runs` batch abandoned after its retry reports zero rows deleted for + `run_events`, `office_run_route_attempts` and `office_run_skills`, not the + counts the rolled-back statements addressed (002.7, 004.6). + +Commands: + +``` +cd apps/backend +go test ./internal/office/... ./internal/runs/... -race -count=1 +go run ./cmd/sqlguard ./internal +go test -race ./internal/persistence/storeconformance -count=1 +KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... \ + ./internal/persistence/storeconformance -count=1 +cd apps/web && pnpm run typecheck && pnpm run i18n:check +``` + +The PostgreSQL line is not optional. The repository's dialect-sensitive suites +self-skip when `KANDEV_TEST_POSTGRES_DSN` is unset, so a green local run without +it proves nothing about AC-OFFICE-RUN-HISTORY-RETENTION-005.1. The engine-parity +test asserts identical deleted sets, retained sets, and counts from identical +seed data on both engines. + +## Rejected alternatives + +- **Wire the existing `CleanExpired` and stop.** One line, and it orphans every + `run_events` row it passes, permanently and invisibly, because no foreign key + on either engine would clean up after it. +- **Add `ON DELETE CASCADE` to the satellite tables instead.** A schema change + to three tables on two engines, requiring a table rebuild on SQLite, to avoid + three `DELETE` statements. It also hides the deletion from the code that has + to count it for the sweep report. +- **Age-prune `run_events` directly.** The obvious implementation, and the one + that breaks the run detail view's incremental tail and can restart a live + run's sequence at zero. Forbidden by 001.6. +- **Run the sweep on the 5s Office tick.** Rejected in 002.1. +- **Count-only retention, keep newest N per owner with no window.** Bounds the + table but makes "how long is my history kept" unanswerable for an install + with mixed routine frequencies. The floor covers the case count-only is good + at; the window covers the case it is bad at. +- **Deletion off by default.** Safe, and it means the gap stays open on every + install that never visits the settings page. The per-table preview, designed + in [run history retention + operations](run-history-retention-operations.md), gives the same protection + without that outcome. +- **Persist sweep history.** Named in the operations requirement's exclusions. + +- **Rely on the window function's `ORDER BY` to order the batch.** It ranks rows + inside each partition; it does not order the rows the outer `LIMIT` draws + from. Leaving the outer select unordered makes a backlog sweep's deleted set + engine-dependent and quietly breaks + AC-OFFICE-RUN-HISTORY-RETENTION-005.1 only on installs large enough to exceed + the batch limit — the installs that need this feature most. + +## Prior art, applied + +**Wiki: unavailable** — receipt in the requirement document. Nothing here should +be read as departing from a wiki position, because none could be consulted. + +**`internal/automation/run_retention.go`** is the closest in-repo precedent, and it +prunes worktrees rather than rows. Three things are taken from it: retention scoped +per owner rather than globally, so one noisy owner cannot evict a quiet one's +only record; a bounded sweep window so a backlog drains across sweeps instead +of walking the whole table each time; and re-checking liveness immediately +before the destructive act, which appears here as the re-asserted predicate and +the batch rollback. One thing is deliberately not taken: it hangs its sweep off +a finalization hook, which ties cleanup frequency to firing frequency and leaves +a stopped routine's history untouched forever. This design uses a clock. + +**`internal/system/storage`** supplies the scheduler shape, the settings +storage and normalization pattern, the hours-based interval with min and max +bounds, and the `health.Checker` route to a production-visible warning. + +**GitLab Duo** and **the Claude apps gateway** are surveyed in the requirement +document's Prior art. What this design takes from them: the 30-day default +window, and the history-vs-live-state split (their per-table windows with one +table marked "until deleted via the API" rather than given a window). Neither +previews before the first deletion; that addition is ours, and it exists because +this ships enabled by default onto installs that already hold history.