diff --git a/.github/workflows/backend-tests.yml b/.github/workflows/backend-tests.yml
index e3d01d15396..1f7650c4711 100644
--- a/.github/workflows/backend-tests.yml
+++ b/.github/workflows/backend-tests.yml
@@ -624,6 +624,7 @@ jobs:
./internal/notifications/store
./internal/office/configsync
./internal/office/repository/sqlite
+ ./internal/office/retention
./internal/orchestrator/messagequeue
./internal/persistence
./internal/persistence/storeconformance
diff --git a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go
index d885a2fcaf3..f1ad97c6317 100644
--- a/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go
+++ b/apps/backend/internal/agent/runtime/lifecycle/manager_subscription_test.go
@@ -269,17 +269,18 @@ func TestAggregator_FailedPausedPushIsRetriedOnUnchangedContribution(t *testing.
w.WriteHeader(http.StatusBadRequest)
return
}
- modes <- body.Mode
if body.Mode == string(WorkspacePollModePaused) {
mu.Lock()
pausedCalls++
call := pausedCalls
mu.Unlock()
if call == 1 {
+ modes <- body.Mode
w.WriteHeader(http.StatusServiceUnavailable)
return
}
}
+ modes <- body.Mode
w.WriteHeader(http.StatusOK)
}))
t.Cleanup(srv.Close)
diff --git a/apps/backend/internal/backendapp/helpers.go b/apps/backend/internal/backendapp/helpers.go
index 859ad57c127..d06ea408750 100644
--- a/apps/backend/internal/backendapp/helpers.go
+++ b/apps/backend/internal/backendapp/helpers.go
@@ -65,6 +65,7 @@ import (
notificationhandlers "github.com/kandev/kandev/internal/notifications/handlers"
officeagents "github.com/kandev/kandev/internal/office/agents"
officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ "github.com/kandev/kandev/internal/office/retention"
officetestharness "github.com/kandev/kandev/internal/office/testharness"
"github.com/kandev/kandev/internal/orchestrator"
"github.com/kandev/kandev/internal/org"
@@ -1521,6 +1522,7 @@ func registerSecondaryRoutes(
registerHealthRoutes(p)
registerSystemRoutes(p)
+ registerRetentionRoutes(p)
if p.runtimeFlagsSvc != nil {
runtimeflags.RegisterRoutes(p.router, p.runtimeFlagsSvc)
}
@@ -1722,6 +1724,23 @@ func registerSystemRoutes(p routeParams) {
p.systemSvc.RegisterRoutes(p.router, p.log)
}
+// registerRetentionRoutes mounts GET/PUT /api/v1/system/retention. It is a
+// separate group from systemSvc's own /api/v1/system group (rather than a
+// field on system.Service) because internal/office/retention cannot be
+// imported by internal/system without inverting the existing system ->
+// office dependency direction; gin allows two RouterGroups to share a path
+// prefix as long as no route collides, and none does here. Read/admin
+// split mirrors system.Service.RegisterRoutes: GET is member-readable,
+// PUT requires the admin-scoped settings-manage permission.
+func registerRetentionRoutes(p routeParams) {
+ if p.services == nil || p.services.Retention == nil {
+ return
+ }
+ read := p.router.Group("/api/v1/system")
+ admin := read.Group("", authz.RequireOrgScope(authz.ScopeOrgSettingsManage))
+ retention.RegisterRoutes(read, admin, p.services.Retention.Handler)
+}
+
// registerHealthRoutes sets up the system health endpoint with all health checkers.
func registerHealthRoutes(p routeParams) {
var githubProvider health.GitHubStatusProvider
@@ -1747,6 +1766,9 @@ func registerHealthRoutes(p routeParams) {
if p.systemSvc != nil && p.systemSvc.StorageRuntime != nil {
checkers = append(checkers, p.systemSvc.StorageRuntime)
}
+ if p.services != nil && p.services.Retention != nil {
+ checkers = append(checkers, p.services.Retention.Checker)
+ }
healthSvc := health.NewService(p.log, checkers...)
health.RegisterRoutes(p.router, healthSvc, p.log)
}
diff --git a/apps/backend/internal/backendapp/main.go b/apps/backend/internal/backendapp/main.go
index 845bc80d67f..7ef9ad79846 100644
--- a/apps/backend/internal/backendapp/main.go
+++ b/apps/backend/internal/backendapp/main.go
@@ -100,6 +100,7 @@ import (
officepause "github.com/kandev/kandev/internal/office/pause"
officeprojects "github.com/kandev/kandev/internal/office/projects"
officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ "github.com/kandev/kandev/internal/office/retention"
officeroutines "github.com/kandev/kandev/internal/office/routines"
"github.com/kandev/kandev/internal/office/routing"
officescheduler "github.com/kandev/kandev/internal/office/scheduler"
@@ -1181,6 +1182,16 @@ func startGatewayAndServe(
})
systemSvc.Storage = storageComposition.handler
systemSvc.StorageRuntime = storageComposition.runtime
+
+ // Office run history retention: bounds office_routine_runs, runs, and
+ // their satellites on its own interval, separate from the 5s Office
+ // tick. Kept regardless of the Office feature flag — see Services.Retention.
+ services.Retention = retention.NewRuntime(dbPool, repos.SystemSettings,
+ func(message string, err error) { log.Error(message, zap.Error(err)) })
+ if err := services.Retention.Start(ctx); err != nil {
+ log.Warn("office run retention scheduler failed to start", zap.Error(err))
+ }
+ addCleanup(func() error { services.Retention.Stop(); return nil })
if systemSvc.LogBundles != nil {
systemSvc.LogBundles.SetNotifier(gateway.Hub)
systemSvc.LogBundles.SetSessionProvider(newDiagnosticSessionProvider(services.Task))
diff --git a/apps/backend/internal/backendapp/types.go b/apps/backend/internal/backendapp/types.go
index 44107a9285f..84878a81fd7 100644
--- a/apps/backend/internal/backendapp/types.go
+++ b/apps/backend/internal/backendapp/types.go
@@ -25,6 +25,7 @@ import (
notificationstore "github.com/kandev/kandev/internal/notifications/store"
office "github.com/kandev/kandev/internal/office"
officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ "github.com/kandev/kandev/internal/office/retention"
officeservice "github.com/kandev/kandev/internal/office/service"
"github.com/kandev/kandev/internal/org"
"github.com/kandev/kandev/internal/orgunit"
@@ -112,6 +113,12 @@ type Services struct {
// WorktreeMgr is the worktree manager. Exposed here so the install-wide
// storage-maintenance composition can reach it for workspace cleanup.
WorktreeMgr *worktree.Manager
+ // Retention owns the office_routine_runs/runs history sweep scheduler,
+ // its HTTP surface, and its health checker. Kept regardless of the
+ // Office feature flag, matching every other required-schema owner: rows
+ // written while Office was enabled still need bounding after it is
+ // turned off.
+ Retention *retention.Runtime
// Terminal is the first-class user-terminal service (rename, park, etc.).
// Wired into the gateway once lifecycle.Manager is up so the PTY backend
// is available.
diff --git a/apps/backend/internal/office/repository/sqlite/base_migrations.go b/apps/backend/internal/office/repository/sqlite/base_migrations.go
index 75f2b6c6e99..968b8e9e7af 100644
--- a/apps/backend/internal/office/repository/sqlite/base_migrations.go
+++ b/apps/backend/internal/office/repository/sqlite/base_migrations.go
@@ -66,6 +66,9 @@ func (r *Repository) runMigrations() error {
r.migrateBudgetPolicyRevision()
r.migrateWorkspacePauseSkipAttribution()
r.migrateLoopLivenessCausationID()
+ if err := r.migrateRetentionIndexes(); err != nil {
+ return err
+ }
if err := r.migrate.Err(); err != nil {
return err
}
@@ -141,6 +144,27 @@ func (r *Repository) migrateLoopLivenessCausationID() {
ON runs(causation_id) WHERE causation_id != ''`)
}
+// migrateRetentionIndexes adds the two indexes the run-history retention
+// sweep depends on (docs/specs/office/system-design/run-history-retention.md
+// "Indexes to add"). Both are expression indexes over the same
+// COALESCE(...) the sweep both filters and orders by; a plain-column index
+// on the nullable completion column would serve neither the WHERE clause
+// nor the ORDER BY the sweep actually issues, on either engine.
+func (r *Repository) migrateRetentionIndexes() error {
+ if err := r.migrate.Apply(
+ "idx_office_routine_runs_retention",
+ `CREATE INDEX IF NOT EXISTS idx_office_routine_runs_retention
+ ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)`,
+ ); err != nil {
+ return err
+ }
+ return r.migrate.Apply(
+ "idx_runs_retention",
+ `CREATE INDEX IF NOT EXISTS idx_runs_retention
+ ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)`,
+ )
+}
+
// migrateContinuationScope adds runs.continuation_scope for databases
// created before WO-16's claim-time scope persistence. Existing rows receive
// a scope from their stored context snapshot so queued or claimed taskless
@@ -418,6 +442,9 @@ func (r *Repository) migrateFailureColumns() error {
if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_agent ON office_agent_pause_recoveries(agent_id)`); err != nil {
return fmt.Errorf("idx_office_agent_pause_recoveries_agent: %w", err)
}
+ if _, err := r.db.Exec(`CREATE INDEX IF NOT EXISTS idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)`); err != nil {
+ return fmt.Errorf("idx_office_agent_pause_recoveries_failed_run: %w", err)
+ }
return nil
}
diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go
new file mode 100644
index 00000000000..dde7028f0f5
--- /dev/null
+++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_postgres_test.go
@@ -0,0 +1,53 @@
+package sqlite_test
+
+import (
+ "testing"
+
+ "github.com/kandev/kandev/internal/office/repository/sqlite"
+ taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite"
+ "github.com/kandev/kandev/internal/testutil"
+)
+
+// TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe is the PostgreSQL
+// half of TestRetentionIndexes_CreatedFreshAndReplaySafe: the two
+// expression indexes the retention sweep depends on must exist there too,
+// with identical CREATE INDEX IF NOT EXISTS replay safety. Skips unless
+// KANDEV_TEST_POSTGRES_DSN is set.
+func TestPostgresRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ conn := testutil.OpenIsolatedPostgres(t, dsn)
+
+ // tasks is created by the task repository's schema init, mirroring
+ // production boot order (see child_summaries_postgres_test.go).
+ if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init task repo: %v", err)
+ }
+ if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("fresh NewWithDB: %v", err)
+ }
+ assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention")
+ assertPostgresIndexExists(t, conn, "idx_runs_retention")
+ assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run")
+
+ if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("replay NewWithDB: %v", err)
+ }
+ assertPostgresIndexExists(t, conn, "idx_office_routine_runs_retention")
+ assertPostgresIndexExists(t, conn, "idx_runs_retention")
+ assertPostgresIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run")
+}
+
+func assertPostgresIndexExists(t *testing.T, conn interface {
+ Get(dest interface{}, query string, args ...interface{}) error
+}, name string) {
+ t.Helper()
+ var count int
+ if err := conn.Get(&count,
+ `SELECT COUNT(*) FROM pg_indexes WHERE indexname = $1`, name,
+ ); err != nil {
+ t.Fatalf("query pg_indexes for %s: %v", name, err)
+ }
+ if count != 1 {
+ t.Fatalf("index %s: found %d, want 1", name, count)
+ }
+}
diff --git a/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go
new file mode 100644
index 00000000000..1ee5fe893d2
--- /dev/null
+++ b/apps/backend/internal/office/repository/sqlite/retention_indexes_test.go
@@ -0,0 +1,50 @@
+package sqlite_test
+
+import (
+ "testing"
+
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/office/repository/sqlite"
+)
+
+// TestRetentionIndexes_CreatedFreshAndReplaySafe proves the retention indexes
+// exist after a fresh boot and that re-running schema init against the same
+// database (the upgrade-path replay) is a no-op, not an error.
+func TestRetentionIndexes_CreatedFreshAndReplaySafe(t *testing.T) {
+ conn, err := sqlx.Open("sqlite3", ":memory:")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+
+ if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("fresh NewWithDB: %v", err)
+ }
+ assertIndexExists(t, conn, "idx_office_routine_runs_retention")
+ assertIndexExists(t, conn, "idx_runs_retention")
+ assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run")
+
+ // Replay: schema init against the same, already-initialized database.
+ if _, err := sqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("replay NewWithDB: %v", err)
+ }
+ assertIndexExists(t, conn, "idx_office_routine_runs_retention")
+ assertIndexExists(t, conn, "idx_runs_retention")
+ assertIndexExists(t, conn, "idx_office_agent_pause_recoveries_failed_run")
+}
+
+func assertIndexExists(t *testing.T, conn *sqlx.DB, name string) {
+ t.Helper()
+ var count int
+ if err := conn.Get(&count,
+ `SELECT COUNT(*) FROM sqlite_master WHERE type = 'index' AND name = ?`, name,
+ ); err != nil {
+ t.Fatalf("query sqlite_master for %s: %v", name, err)
+ }
+ if count != 1 {
+ t.Fatalf("index %s: found %d, want 1", name, count)
+ }
+}
diff --git a/apps/backend/internal/office/retention/census.go b/apps/backend/internal/office/retention/census.go
new file mode 100644
index 00000000000..061f66f2561
--- /dev/null
+++ b/apps/backend/internal/office/retention/census.go
@@ -0,0 +1,170 @@
+package retention
+
+import (
+ "encoding/json"
+ "fmt"
+ "sort"
+ "sync"
+ "time"
+)
+
+// CensusState is the tri-state freshness of a thresholded table's retained
+// count (AC-OFFICE-RUN-HISTORY-RETENTION-003.11): zero is a real
+// measurement, so "nobody has counted yet" cannot be represented as a
+// count of zero.
+type CensusState int
+
+const (
+ // CensusNotComputed is the state before the first evaluation ever
+ // succeeds for a table. AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and
+ // -004.8 both require the surface to render this honestly rather than
+ // as a zero count.
+ CensusNotComputed CensusState = iota
+ // CensusFresh means RetainedCount reflects the most recent evaluation,
+ // which succeeded.
+ CensusFresh
+ // CensusStale means the most recent evaluation failed, and the fields
+ // below are carried over unchanged from the last one that succeeded
+ // (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's "keeps the last successful
+ // counts, reports them stale" — extended to the unknown-status warning
+ // riding the same query, since neither the requirement nor the design
+ // says the two field groups should diverge on a failed evaluation).
+ CensusStale
+)
+
+var censusStateNames = map[CensusState]string{
+ CensusNotComputed: "not_computed",
+ CensusFresh: "fresh",
+ CensusStale: "stale",
+}
+
+// MarshalJSON renders the state as its stable wire name rather than the
+// underlying int, so an HTTP consumer never has to hardcode 0/1/2.
+func (s CensusState) MarshalJSON() ([]byte, error) {
+ name, ok := censusStateNames[s]
+ if !ok {
+ return nil, fmt.Errorf("retention: unknown census state %d", s)
+ }
+ return json.Marshal(name)
+}
+
+// UnmarshalJSON accepts only the names MarshalJSON produces.
+func (s *CensusState) UnmarshalJSON(data []byte) error {
+ var name string
+ if err := json.Unmarshal(data, &name); err != nil {
+ return err
+ }
+ for state, candidate := range censusStateNames {
+ if candidate == name {
+ *s = state
+ return nil
+ }
+ }
+ return fmt.Errorf("retention: unknown census state %q", name)
+}
+
+// UnknownStatusCount is one status this package does not recognize, with
+// the number of retained rows currently holding it
+// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10).
+type UnknownStatusCount struct {
+ Status string `json:"status"`
+ Count int64 `json:"count"`
+}
+
+// TableCensus is one thresholded table's retained-count evaluation.
+type TableCensus struct {
+ State CensusState `json:"state"`
+ RetainedCount int64 `json:"retained_count"`
+ AsOf time.Time `json:"as_of"`
+ UnknownStatuses []UnknownStatusCount `json:"unknown_statuses,omitempty"` // ascending by status; only populated for status-bearing tables
+ TopRoutineID string `json:"top_routine_id,omitempty"` // office_routine_runs only; empty when not applicable
+ TopRoutineShare float64 `json:"top_routine_share,omitempty"` // top routine's retained rows / table's retained count
+}
+
+// RetainedCounts holds the current census result for every thresholded
+// table (AC-OFFICE-RUN-HISTORY-RETENTION-003.11).
+type RetainedCounts struct {
+ OfficeRoutineRuns TableCensus `json:"office_routine_runs"`
+ Runs TableCensus `json:"runs"`
+ RunEvents TableCensus `json:"run_events"`
+}
+
+// summarizeStatusCensus turns a status->count breakdown into a retained
+// count (the sum across every status, since "retained" is the table's
+// current row count) and the status/count list, sorted ascending by
+// status, for statuses belonging to neither the history nor the
+// live-state set (AC-OFFICE-RUN-HISTORY-RETENTION-001.10).
+func summarizeStatusCensus(counts map[string]int64, history, live []string) (retained int64, unknown []UnknownStatusCount) {
+ known := make(map[string]bool, len(history)+len(live))
+ for _, s := range history {
+ known[s] = true
+ }
+ for _, s := range live {
+ known[s] = true
+ }
+ for status, count := range counts {
+ retained += count
+ if !known[status] {
+ unknown = append(unknown, UnknownStatusCount{Status: status, Count: count})
+ }
+ }
+ sort.Slice(unknown, func(i, j int) bool { return unknown[i].Status < unknown[j].Status })
+ return retained, unknown
+}
+
+// CensusTracker holds the latest RetainedCounts in memory, applying the
+// tri-state rule per table: a successful evaluation replaces a table's
+// entry and marks it fresh; a failed evaluation leaves an already-computed
+// entry in place and marks it stale, or leaves a never-computed entry as
+// not-yet-computed. One table's failure never touches another table's
+// entry. Safe for concurrent use: Record* is called from the scheduler
+// goroutine and Snapshot from HTTP handlers.
+type CensusTracker struct {
+ mu sync.Mutex
+ counts RetainedCounts
+}
+
+// NewCensusTracker returns a tracker with every table not yet computed.
+func NewCensusTracker() *CensusTracker {
+ return &CensusTracker{}
+}
+
+// Snapshot returns the current RetainedCounts.
+func (t *CensusTracker) Snapshot() RetainedCounts {
+ t.mu.Lock()
+ defer t.mu.Unlock()
+ return t.counts
+}
+
+// RecordRoutineRuns applies an office_routine_runs evaluation outcome.
+func (t *CensusTracker) RecordRoutineRuns(fresh TableCensus, err error) {
+ t.mu.Lock()
+ defer t.mu.Unlock()
+ t.counts.OfficeRoutineRuns = applyCensusResult(t.counts.OfficeRoutineRuns, fresh, err)
+}
+
+// RecordRuns applies a runs evaluation outcome.
+func (t *CensusTracker) RecordRuns(fresh TableCensus, err error) {
+ t.mu.Lock()
+ defer t.mu.Unlock()
+ t.counts.Runs = applyCensusResult(t.counts.Runs, fresh, err)
+}
+
+// RecordRunEvents applies a run_events evaluation outcome.
+func (t *CensusTracker) RecordRunEvents(fresh TableCensus, err error) {
+ t.mu.Lock()
+ defer t.mu.Unlock()
+ t.counts.RunEvents = applyCensusResult(t.counts.RunEvents, fresh, err)
+}
+
+func applyCensusResult(prev, fresh TableCensus, err error) TableCensus {
+ if err != nil {
+ if prev.State == CensusNotComputed {
+ return prev
+ }
+ prev.State = CensusStale
+ return prev
+ }
+ fresh.State = CensusFresh
+ return fresh
+}
diff --git a/apps/backend/internal/office/retention/census_json_test.go b/apps/backend/internal/office/retention/census_json_test.go
new file mode 100644
index 00000000000..7590ccf1ca5
--- /dev/null
+++ b/apps/backend/internal/office/retention/census_json_test.go
@@ -0,0 +1,101 @@
+package retention
+
+import (
+ "encoding/json"
+ "testing"
+)
+
+func TestCensusState_MarshalJSON_UsesStableNames(t *testing.T) {
+ cases := map[CensusState]string{
+ CensusNotComputed: `"not_computed"`,
+ CensusFresh: `"fresh"`,
+ CensusStale: `"stale"`,
+ }
+ for state, want := range cases {
+ got, err := json.Marshal(state)
+ if err != nil {
+ t.Fatalf("Marshal(%v): %v", state, err)
+ }
+ if string(got) != want {
+ t.Fatalf("Marshal(%v) = %s, want %s", state, got, want)
+ }
+ }
+}
+
+func TestCensusState_UnmarshalJSON_RoundTrips(t *testing.T) {
+ for _, state := range []CensusState{CensusNotComputed, CensusFresh, CensusStale} {
+ encoded, err := json.Marshal(state)
+ if err != nil {
+ t.Fatalf("Marshal(%v): %v", state, err)
+ }
+ var decoded CensusState
+ if err := json.Unmarshal(encoded, &decoded); err != nil {
+ t.Fatalf("Unmarshal(%s): %v", encoded, err)
+ }
+ if decoded != state {
+ t.Fatalf("round trip %v -> %s -> %v", state, encoded, decoded)
+ }
+ }
+}
+
+func TestCensusState_UnmarshalJSON_RejectsUnknownName(t *testing.T) {
+ var state CensusState
+ if err := json.Unmarshal([]byte(`"bogus"`), &state); err == nil {
+ t.Fatal("expected an error for an unrecognized census state name")
+ }
+}
+
+func TestRetainedCounts_JSONUsesSnakeCaseFieldNames(t *testing.T) {
+ counts := RetainedCounts{
+ OfficeRoutineRuns: TableCensus{State: CensusFresh, RetainedCount: 3},
+ }
+ encoded, err := json.Marshal(counts)
+ if err != nil {
+ t.Fatalf("Marshal: %v", err)
+ }
+ var raw map[string]json.RawMessage
+ if err := json.Unmarshal(encoded, &raw); err != nil {
+ t.Fatalf("Unmarshal into map: %v", err)
+ }
+ for _, key := range []string{"office_routine_runs", "runs", "run_events"} {
+ if _, ok := raw[key]; !ok {
+ t.Fatalf("RetainedCounts JSON missing key %q; got %s", key, encoded)
+ }
+ }
+}
+
+func TestLastSweep_JSONUsesSnakeCaseFieldNames(t *testing.T) {
+ sweep := LastSweep{
+ OfficeRoutineRuns: SweptTableResult{
+ TableSweepResult: TableSweepResult{Deleted: 5, Backlog: true},
+ Previewed: true,
+ WouldDelete: 7,
+ },
+ }
+ encoded, err := json.Marshal(sweep)
+ if err != nil {
+ t.Fatalf("Marshal: %v", err)
+ }
+ var raw map[string]json.RawMessage
+ if err := json.Unmarshal(encoded, &raw); err != nil {
+ t.Fatalf("Unmarshal into map: %v", err)
+ }
+ for _, key := range []string{
+ "started_at", "finished_at", "office_routine_runs", "runs",
+ "run_events", "route_attempts", "run_skills",
+ } {
+ if _, ok := raw[key]; !ok {
+ t.Fatalf("LastSweep JSON missing key %q; got %s", key, encoded)
+ }
+ }
+
+ var routineRuns map[string]json.RawMessage
+ if err := json.Unmarshal(raw["office_routine_runs"], &routineRuns); err != nil {
+ t.Fatalf("Unmarshal office_routine_runs: %v", err)
+ }
+ for _, key := range []string{"deleted", "backlog", "error", "previewed", "would_delete"} {
+ if _, ok := routineRuns[key]; !ok {
+ t.Fatalf("SweptTableResult JSON missing key %q; got %s", key, raw["office_routine_runs"])
+ }
+ }
+}
diff --git a/apps/backend/internal/office/retention/census_test.go b/apps/backend/internal/office/retention/census_test.go
new file mode 100644
index 00000000000..c776d5bf5e3
--- /dev/null
+++ b/apps/backend/internal/office/retention/census_test.go
@@ -0,0 +1,133 @@
+package retention
+
+import (
+ "errors"
+ "reflect"
+ "testing"
+ "time"
+)
+
+func TestSummarizeStatusCensus_SumsAllStatusesRegardlessOfClass(t *testing.T) {
+ counts := map[string]int64{
+ "done": 3,
+ "received": 2,
+ }
+ retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses)
+ if retained != 5 {
+ t.Fatalf("retained = %d, want 5", retained)
+ }
+ if len(unknown) != 0 {
+ t.Fatalf("unknown = %v, want none", unknown)
+ }
+}
+
+func TestSummarizeStatusCensus_DetectsUnknownStatusesSortedAscending(t *testing.T) {
+ counts := map[string]int64{
+ "done": 1,
+ "zeta": 1,
+ "alpha": 1,
+ "skipped": 1,
+ }
+ retained, unknown := summarizeStatusCensus(counts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses)
+ if retained != 4 {
+ t.Fatalf("retained = %d, want 4", retained)
+ }
+ want := []UnknownStatusCount{{Status: "alpha", Count: 1}, {Status: "zeta", Count: 1}}
+ if !reflect.DeepEqual(unknown, want) {
+ t.Fatalf("unknown = %v, want %v", unknown, want)
+ }
+}
+
+func TestSummarizeStatusCensus_EmptyTableIsZeroNotUnknown(t *testing.T) {
+ retained, unknown := summarizeStatusCensus(map[string]int64{}, RunHistoryStatuses, RunLiveStatuses)
+ if retained != 0 {
+ t.Fatalf("retained = %d, want 0", retained)
+ }
+ if unknown != nil {
+ t.Fatalf("unknown = %v, want nil", unknown)
+ }
+}
+
+func TestCensusTracker_SnapshotStartsNotComputedForEveryTable(t *testing.T) {
+ tracker := NewCensusTracker()
+ snap := tracker.Snapshot()
+ for name, c := range map[string]TableCensus{
+ "office_routine_runs": snap.OfficeRoutineRuns,
+ "runs": snap.Runs,
+ "run_events": snap.RunEvents,
+ } {
+ if c.State != CensusNotComputed {
+ t.Fatalf("%s: state = %v, want CensusNotComputed", name, c.State)
+ }
+ }
+}
+
+func TestCensusTracker_SuccessfulEvaluationIsFresh(t *testing.T) {
+ tracker := NewCensusTracker()
+ now := time.Now().UTC()
+ tracker.RecordRuns(TableCensus{RetainedCount: 42, AsOf: now}, nil)
+
+ got := tracker.Snapshot().Runs
+ if got.State != CensusFresh {
+ t.Fatalf("state = %v, want CensusFresh", got.State)
+ }
+ if got.RetainedCount != 42 {
+ t.Fatalf("retained = %d, want 42", got.RetainedCount)
+ }
+ if !got.AsOf.Equal(now) {
+ t.Fatalf("asOf = %v, want %v", got.AsOf, now)
+ }
+}
+
+func TestCensusTracker_FailedEvaluationWithNoPriorSuccessStaysNotComputed(t *testing.T) {
+ tracker := NewCensusTracker()
+ tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom"))
+
+ got := tracker.Snapshot().OfficeRoutineRuns
+ if got.State != CensusNotComputed {
+ t.Fatalf("state = %v, want CensusNotComputed", got.State)
+ }
+}
+
+func TestCensusTracker_FailedEvaluationAfterSuccessKeepsLastCountsMarkedStale(t *testing.T) {
+ tracker := NewCensusTracker()
+ first := TableCensus{
+ RetainedCount: 100,
+ AsOf: time.Now().UTC(),
+ UnknownStatuses: []UnknownStatusCount{{Status: "weird", Count: 1}},
+ TopRoutineID: "r-1",
+ TopRoutineShare: 0.5,
+ }
+ tracker.RecordRoutineRuns(first, nil)
+
+ tracker.RecordRoutineRuns(TableCensus{}, errors.New("query failed"))
+
+ got := tracker.Snapshot().OfficeRoutineRuns
+ if got.State != CensusStale {
+ t.Fatalf("state = %v, want CensusStale", got.State)
+ }
+ if got.RetainedCount != 100 {
+ t.Fatalf("retained = %d, want 100 (carried over)", got.RetainedCount)
+ }
+ if !reflect.DeepEqual(got.UnknownStatuses, []UnknownStatusCount{{Status: "weird", Count: 1}}) {
+ t.Fatalf("unknownStatuses = %v, want carried over", got.UnknownStatuses)
+ }
+ if got.TopRoutineID != "r-1" || got.TopRoutineShare != 0.5 {
+ t.Fatalf("top routine attribution not carried over: %+v", got)
+ }
+}
+
+func TestCensusTracker_OneTableFailureDoesNotTouchAnother(t *testing.T) {
+ tracker := NewCensusTracker()
+ tracker.RecordRuns(TableCensus{RetainedCount: 7, AsOf: time.Now().UTC()}, nil)
+
+ tracker.RecordRoutineRuns(TableCensus{}, errors.New("boom"))
+
+ snap := tracker.Snapshot()
+ if snap.Runs.State != CensusFresh || snap.Runs.RetainedCount != 7 {
+ t.Fatalf("runs entry disturbed by routine_runs failure: %+v", snap.Runs)
+ }
+ if snap.OfficeRoutineRuns.State != CensusNotComputed {
+ t.Fatalf("routine_runs state = %v, want CensusNotComputed", snap.OfficeRoutineRuns.State)
+ }
+}
diff --git a/apps/backend/internal/office/retention/goleak_test.go b/apps/backend/internal/office/retention/goleak_test.go
new file mode 100644
index 00000000000..7cfa26b30a1
--- /dev/null
+++ b/apps/backend/internal/office/retention/goleak_test.go
@@ -0,0 +1,15 @@
+package retention
+
+import (
+ "testing"
+
+ "go.uber.org/goleak"
+)
+
+// TestMain enforces no goroutine leaks across the retention package.
+// Scheduler.Start spawns a single lifecycle-managed loop goroutine; Stop
+// cancels its context and waits on the WaitGroup. Regressions where Stop
+// forgets to cancel or a test leaves a scheduler running surface here.
+func TestMain(m *testing.M) {
+ goleak.VerifyTestMain(m)
+}
diff --git a/apps/backend/internal/office/retention/handler.go b/apps/backend/internal/office/retention/handler.go
new file mode 100644
index 00000000000..d68b5e83826
--- /dev/null
+++ b/apps/backend/internal/office/retention/handler.go
@@ -0,0 +1,141 @@
+package retention
+
+import (
+ "errors"
+ "net/http"
+ "sync"
+ "time"
+
+ "github.com/gin-gonic/gin"
+)
+
+const responseErrorKey = "error"
+
+// maxRetentionSettingsBodyBytes bounds administrator-controlled JSON before
+// the handler decodes it, so a malformed request cannot consume unbounded
+// memory.
+const maxRetentionSettingsBodyBytes = 1 << 20
+
+// HandlerConfig wires the HTTP surface to the package's own stores.
+type HandlerConfig struct {
+ SettingsStore *SettingsStore
+ Sweeper *Sweeper
+ // OnSettingsChanged, when set, is called with the normalized document
+ // after a successful PUT so the running scheduler re-arms its timers
+ // from the new settings without a restart (AC-004.5).
+ OnSettingsChanged func(Settings)
+ LogError func(string, error)
+}
+
+// Handler serves GET/PUT /api/v1/system/retention.
+type Handler struct {
+ config HandlerConfig
+
+ // mu serializes a PUT's save and scheduler-apply as one critical
+ // section, so two concurrent PUTs cannot interleave into the scheduler
+ // applying the older of the two writes after the newer one is already
+ // stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.5's last-writer-wins).
+ mu sync.Mutex
+}
+
+// NewHandler wires a Handler to its dependencies.
+func NewHandler(config HandlerConfig) *Handler {
+ return &Handler{config: config}
+}
+
+func (h *Handler) logError(message string, err error) {
+ if h.config.LogError != nil {
+ h.config.LogError(message, err)
+ }
+}
+
+// RegisterRoutes wires GET/PUT /api/v1/system/retention: GET is
+// member-readable, matching every sibling System-pages surface (storage,
+// queue settings, sleep inhibition), and readable while retention is
+// disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8); PUT is admin-scoped.
+func RegisterRoutes(read, admin *gin.RouterGroup, handler *Handler) {
+ read.GET("/retention", handler.getRetention)
+ admin.PUT("/retention", handler.putRetention)
+}
+
+// Status is the GET response body: effective settings, the most recently
+// completed sweep (nil before the first one — AC-004.7), the
+// separately-held skip record, and the per-thresholded-table retained-count
+// census (AC-004.6).
+type Status struct {
+ Settings Settings `json:"settings"`
+ LastSweep *LastSweep `json:"last_sweep"`
+ SkipCount int64 `json:"skip_count"`
+ LastSkipAt *time.Time `json:"last_skip_at,omitempty"`
+ RetainedCounts RetainedCounts `json:"retained_counts"`
+}
+
+func (h *Handler) getRetention(c *gin.Context) {
+ settings, err := h.config.SettingsStore.GetSettings(c.Request.Context())
+ if err != nil {
+ h.logError("failed to load retention settings", err)
+ }
+
+ status := Status{
+ Settings: settings,
+ RetainedCounts: h.config.Sweeper.CensusSnapshot(),
+ }
+ if last, ok := h.config.Sweeper.LastSweepSnapshot(); ok {
+ status.LastSweep = &last
+ }
+ if count, lastAt := h.config.Sweeper.SkipSnapshot(); count > 0 {
+ status.SkipCount = count
+ status.LastSkipAt = &lastAt
+ }
+ c.JSON(http.StatusOK, status)
+}
+
+func (h *Handler) putRetention(c *gin.Context) {
+ c.Request.Body = http.MaxBytesReader(c.Writer, c.Request.Body, maxRetentionSettingsBodyBytes)
+ body, err := c.GetRawData()
+ if err != nil {
+ var maxBytesErr *http.MaxBytesError
+ if errors.As(err, &maxBytesErr) {
+ c.JSON(http.StatusRequestEntityTooLarge, gin.H{responseErrorKey: "request body too large"})
+ return
+ }
+ c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: "failed to read request body"})
+ return
+ }
+
+ settings, err := decodeRetentionSettings(body)
+ if err != nil {
+ c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()})
+ return
+ }
+
+ h.mu.Lock()
+ defer h.mu.Unlock()
+
+ saved, err := h.config.SettingsStore.SaveSettings(c.Request.Context(), settings)
+ if err != nil {
+ if errors.Is(err, ErrValidation) {
+ c.JSON(http.StatusBadRequest, gin.H{responseErrorKey: err.Error()})
+ return
+ }
+ h.logError("failed to save retention settings", err)
+ c.JSON(http.StatusInternalServerError, gin.H{responseErrorKey: "failed to save retention settings"})
+ return
+ }
+
+ if testBetweenSaveAndApply != nil {
+ testBetweenSaveAndApply()
+ }
+
+ if h.config.OnSettingsChanged != nil {
+ h.config.OnSettingsChanged(saved)
+ }
+ c.JSON(http.StatusOK, saved)
+}
+
+// testBetweenSaveAndApply, when set, runs after a PUT's SaveSettings
+// commits and before OnSettingsChanged is invoked, while mu is still held —
+// a deterministic seam for proving a second PUT cannot save and apply in
+// between (the concurrent-PUT desync this mutex exists to prevent). Never
+// set outside tests.
+var testBetweenSaveAndApply func()
diff --git a/apps/backend/internal/office/retention/handler_test.go b/apps/backend/internal/office/retention/handler_test.go
new file mode 100644
index 00000000000..a8909ef9b7b
--- /dev/null
+++ b/apps/backend/internal/office/retention/handler_test.go
@@ -0,0 +1,442 @@
+package retention
+
+import (
+ "bytes"
+ "encoding/json"
+ "net/http"
+ "net/http/httptest"
+ "strings"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/gin-gonic/gin"
+
+ "github.com/kandev/kandev/internal/auth/authn"
+)
+
+// newTestRetentionRouter mirrors production wiring: GET is member-readable,
+// PUT requires admin, matching storage's and sleep-inhibition's split
+// read/admin groups.
+func newTestRetentionRouter(handler *Handler) *gin.Engine {
+ return newTestRetentionRouterAs(handler, authn.RoleAdmin)
+}
+
+func newTestRetentionRouterAs(handler *Handler, role authn.Role) *gin.Engine {
+ router := gin.New()
+ router.Use(func(c *gin.Context) {
+ authn.SetOnGin(c, authn.Identity{UserID: "user-1", Role: role})
+ c.Next()
+ })
+ read := router.Group("/api/v1/system")
+ admin := read.Group("", authn.RequireAdmin())
+ RegisterRoutes(read, admin, handler)
+ return router
+}
+
+func newTestHandler(t *testing.T) (*Handler, *Sweeper) {
+ t.Helper()
+ sweeper, _ := newTestSweeper(t)
+ handler := NewHandler(HandlerConfig{SettingsStore: sweeper.settingsStore, Sweeper: sweeper})
+ return handler, sweeper
+}
+
+func doRequest(router *gin.Engine, method, path string, body []byte) *httptest.ResponseRecorder {
+ var reader *bytes.Reader
+ if body != nil {
+ reader = bytes.NewReader(body)
+ } else {
+ reader = bytes.NewReader(nil)
+ }
+ request := httptest.NewRequest(method, path, reader)
+ request.Header.Set("Content-Type", "application/json")
+ response := httptest.NewRecorder()
+ router.ServeHTTP(response, request)
+ return response
+}
+
+func TestGetRetention_FreshInstallReturnsDefaultsAndNilLastSweep(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil)
+ if response.Code != http.StatusOK {
+ t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+
+ var status Status
+ if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil {
+ t.Fatalf("unmarshal: %v", err)
+ }
+ if status.LastSweep != nil {
+ t.Fatalf("LastSweep = %+v, want nil before the first sweep (AC-004.7)", status.LastSweep)
+ }
+ if status.Settings != DefaultSettings() {
+ t.Fatalf("Settings = %+v, want defaults", status.Settings)
+ }
+ if status.SkipCount != 0 {
+ t.Fatalf("SkipCount = %d, want 0", status.SkipCount)
+ }
+}
+
+func TestGetRetention_NonAdminMemberCanRead(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouterAs(handler, authn.RoleMember)
+
+ response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil)
+ if response.Code != http.StatusOK {
+ t.Fatalf("member GET status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+}
+
+func TestPutRetention_NonAdminMemberIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouterAs(handler, authn.RoleMember)
+
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{}`))
+ if response.Code != http.StatusForbidden {
+ t.Fatalf("member PUT status = %d, want 403: %s", response.Code, response.Body.String())
+ }
+}
+
+func TestGetRetention_ReflectsLastSweepAndCensus(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+ ctx := t.Context()
+
+ sweeper.RunCensus(ctx)
+ sweeper.RunSweep(ctx) // preview pass; still sets LastSweep
+
+ response := doRequest(router, http.MethodGet, "/api/v1/system/retention", nil)
+ if response.Code != http.StatusOK {
+ t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+
+ var status Status
+ if err := json.Unmarshal(response.Body.Bytes(), &status); err != nil {
+ t.Fatalf("unmarshal: %v", err)
+ }
+ if status.LastSweep == nil {
+ t.Fatal("LastSweep = nil, want non-nil after a sweep has run")
+ }
+ if status.RetainedCounts.OfficeRoutineRuns.State != CensusFresh {
+ t.Fatalf("RetainedCounts.OfficeRoutineRuns.State = %v, want fresh", status.RetainedCounts.OfficeRoutineRuns.State)
+ }
+}
+
+func TestPutRetention_OmittedFieldTakesDocumentedDefault(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"enabled": false}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusOK {
+ t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+
+ var saved Settings
+ if err := json.Unmarshal(response.Body.Bytes(), &saved); err != nil {
+ t.Fatalf("unmarshal: %v", err)
+ }
+ if saved.Enabled {
+ t.Fatal("Enabled = true, want false (explicitly set)")
+ }
+ want := DefaultSettings()
+ if saved.SweepIntervalHours != want.SweepIntervalHours {
+ t.Fatalf("SweepIntervalHours = %d, want the default %d (omitted field)", saved.SweepIntervalHours, want.SweepIntervalHours)
+ }
+ if saved.RoutineRuns != want.RoutineRuns {
+ t.Fatalf("RoutineRuns = %+v, want the default %+v (omitted field)", saved.RoutineRuns, want.RoutineRuns)
+ }
+}
+
+func TestPutRetention_RepeatedIdenticalWriteIsANoOp(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body, err := json.Marshal(DefaultSettings())
+ if err != nil {
+ t.Fatalf("marshal: %v", err)
+ }
+
+ first := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ second := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if first.Code != http.StatusOK || second.Code != http.StatusOK {
+ t.Fatalf("status = %d, %d, want 200, 200", first.Code, second.Code)
+ }
+ if first.Body.String() != second.Body.String() {
+ t.Fatalf("identical writes returned different documents:\n%s\n%s", first.Body.String(), second.Body.String())
+ }
+
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if stored != DefaultSettings() {
+ t.Fatalf("stored = %+v, want unchanged defaults", stored)
+ }
+}
+
+func TestPutRetention_ExplicitNullIsRejectedNamingTheField(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"runs": {"window_days": null}}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ if !strings.Contains(response.Body.String(), "runs.window_days") {
+ t.Fatalf("body = %s, want it to name runs.window_days", response.Body.String())
+ }
+
+ // Nothing written: a decade-long window must survive a null-rejected PUT.
+ settings := DefaultSettings()
+ settings.Runs.WindowDays = 3650
+ if _, err := sweeper.settingsStore.SaveSettings(t.Context(), settings); err != nil {
+ t.Fatalf("seed SaveSettings: %v", err)
+ }
+ response = doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400", response.Code)
+ }
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if stored.Runs.WindowDays != 3650 {
+ t.Fatalf("Runs.WindowDays = %d, want 3650 unchanged (a rejected write must not destroy configured history)", stored.Runs.WindowDays)
+ }
+}
+
+func TestPutRetention_UnknownFieldIsRejectedNamingTheField(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"windw_days": 30}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ if !strings.Contains(response.Body.String(), "windw_days") {
+ t.Fatalf("body = %s, want it to name the misspelled field", response.Body.String())
+ }
+}
+
+func TestPutRetention_TrailingDataAfterObjectIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"enabled": true} {"enabled": false}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if stored != DefaultSettings() {
+ t.Fatalf("stored = %+v, want unchanged defaults (nothing written on rejection)", stored)
+ }
+}
+
+func TestPutRetention_TopLevelNullIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte("null"))
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if stored != DefaultSettings() {
+ t.Fatalf("stored = %+v, want unchanged defaults", stored)
+ }
+}
+
+func TestPutRetention_OversizedBodyIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(strings.Repeat(" ", maxRetentionSettingsBodyBytes+1))
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusRequestEntityTooLarge {
+ t.Fatalf("status = %d, want 413: %s", response.Code, response.Body.String())
+ }
+}
+
+func TestPutRetention_FractionalNumberIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"batch_limit": 100.5}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ if !strings.Contains(response.Body.String(), "batch_limit") {
+ t.Fatalf("body = %s, want it to name batch_limit", response.Body.String())
+ }
+}
+
+func TestPutRetention_WrongTypeIsRejected(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, _ := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"enabled": "yes"}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ if !strings.Contains(response.Body.String(), "enabled") {
+ t.Fatalf("body = %s, want it to name enabled", response.Body.String())
+ }
+}
+
+func TestPutRetention_OutOfRangeIsRejectedAndLeavesStoredSettingsUnchanged(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ handler, sweeper := newTestHandler(t)
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"batch_limit": 1}`) // below minBatchLimit=100
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusBadRequest {
+ t.Fatalf("status = %d, want 400: %s", response.Code, response.Body.String())
+ }
+ if !strings.Contains(response.Body.String(), "batch_limit") {
+ t.Fatalf("body = %s, want it to name batch_limit", response.Body.String())
+ }
+
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if stored != DefaultSettings() {
+ t.Fatalf("stored = %+v, want unchanged defaults", stored)
+ }
+}
+
+func TestPutRetention_SuccessInvokesOnSettingsChanged(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ sweeper, _ := newTestSweeper(t)
+
+ var got Settings
+ var called bool
+ handler := NewHandler(HandlerConfig{
+ SettingsStore: sweeper.settingsStore,
+ Sweeper: sweeper,
+ OnSettingsChanged: func(s Settings) {
+ called = true
+ got = s
+ },
+ })
+ router := newTestRetentionRouter(handler)
+
+ body := []byte(`{"enabled": false}`)
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", body)
+ if response.Code != http.StatusOK {
+ t.Fatalf("status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+ if !called {
+ t.Fatal("OnSettingsChanged was not called")
+ }
+ if got.Enabled {
+ t.Fatal("OnSettingsChanged received Enabled=true, want false")
+ }
+}
+
+// TestPutRetention_ConcurrentPUTsApplyInSaveOrder is the regression test for
+// the concurrent-PUT scheduler desync found in review: without serializing
+// SaveSettings and OnSettingsChanged as one critical section, a second PUT
+// racing between the first's save and apply could complete its own save and
+// apply entirely in between, leaving the scheduler applying the first PUT's
+// now-stale settings after the second PUT's newer write already committed
+// (AC-004.5's last-writer-wins). testBetweenSaveAndApply fires while the
+// first PUT still holds the handler's mutex; it starts a second PUT
+// concurrently and proves that second PUT cannot complete until the first
+// releases the mutex, so the two applications can never interleave.
+func TestPutRetention_ConcurrentPUTsApplyInSaveOrder(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ sweeper, _ := newTestSweeper(t)
+
+ var mu sync.Mutex
+ var applied []bool
+ handler := NewHandler(HandlerConfig{
+ SettingsStore: sweeper.settingsStore,
+ Sweeper: sweeper,
+ OnSettingsChanged: func(s Settings) {
+ mu.Lock()
+ applied = append(applied, s.Enabled)
+ mu.Unlock()
+ },
+ })
+ router := newTestRetentionRouter(handler)
+
+ secondDone := make(chan struct{})
+ secondStarted := false
+
+ t.Cleanup(func() { testBetweenSaveAndApply = nil })
+ testBetweenSaveAndApply = func() {
+ testBetweenSaveAndApply = nil // only race a second request once
+ secondStarted = true
+ go func() {
+ response := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": true}`))
+ if response.Code != http.StatusOK {
+ t.Errorf("second PUT status = %d, want 200: %s", response.Code, response.Body.String())
+ }
+ close(secondDone)
+ }()
+
+ select {
+ case <-secondDone:
+ t.Fatal("second PUT completed while the first still held the critical section")
+ case <-time.After(100 * time.Millisecond):
+ }
+ }
+
+ first := doRequest(router, http.MethodPut, "/api/v1/system/retention", []byte(`{"enabled": false}`))
+ if first.Code != http.StatusOK {
+ t.Fatalf("first PUT status = %d, want 200: %s", first.Code, first.Body.String())
+ }
+ if !secondStarted {
+ t.Fatal("test hook never fired; the race was not exercised")
+ }
+
+ select {
+ case <-secondDone:
+ case <-time.After(5 * time.Second):
+ t.Fatal("second PUT never completed after the first released the critical section")
+ }
+
+ mu.Lock()
+ defer mu.Unlock()
+ if len(applied) != 2 || applied[0] != false || applied[1] != true {
+ t.Fatalf("OnSettingsChanged calls = %+v, want [false, true] in save order", applied)
+ }
+
+ stored, err := sweeper.settingsStore.GetSettings(t.Context())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if !stored.Enabled {
+ t.Fatal("stored Enabled = false, want true (the second, later PUT must win)")
+ }
+}
diff --git a/apps/backend/internal/office/retention/health.go b/apps/backend/internal/office/retention/health.go
new file mode 100644
index 00000000000..67fd613f5d4
--- /dev/null
+++ b/apps/backend/internal/office/retention/health.go
@@ -0,0 +1,240 @@
+package retention
+
+import (
+ "context"
+ "fmt"
+ "sort"
+ "strings"
+ "time"
+
+ "github.com/kandev/kandev/internal/health"
+)
+
+const (
+ fixURL = "/settings/system/data-storage"
+ fixLabel = "Review retention settings"
+)
+
+// Checker implements health.Checker for office run history retention. Every
+// issue is derived fresh from live state on each Check() — LastSweep and
+// RetainedCounts are already the durable, tri-state views this design
+// specifies (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7, -003.11) — so
+// there is no separate stored issue map to keep in sync with them.
+//
+// office_retention_count_failed:
is a ninth issue id beyond the
+// design's closed eight-id catalogue, covering a failed census evaluation.
+// It fires only on CensusStale (a table that had a successful evaluation and
+// then failed); CensusNotComputed is the pre-first-success state AC-003.11
+// requires rendering as absent rather than alarming, so it raises nothing on
+// its own.
+//
+// office_retention_threshold:
and office_retention_disabled:
+// both answer AC-003.5/-003.7's "retained count over threshold" condition,
+// split by whether retention is enabled: AC-003.7 requires the disabled case
+// to additionally state that retention is disabled, and the catalogue gives
+// it its own id rather than a variable message under one id.
+type Checker struct {
+ settingsStore *SettingsStore
+ sweeper *Sweeper
+ previewMarker *PreviewMarkerStore
+}
+
+// NewChecker wires the health checker to the package's own stores.
+func NewChecker(settingsStore *SettingsStore, sweeper *Sweeper, previewMarker *PreviewMarkerStore) *Checker {
+ return &Checker{settingsStore: settingsStore, sweeper: sweeper, previewMarker: previewMarker}
+}
+
+func (c *Checker) Name() string { return "Office run retention" }
+func (c *Checker) Category() string { return "office" }
+
+func (c *Checker) Check(ctx context.Context) []health.Issue {
+ var issues []health.Issue
+
+ settings, err := c.settingsStore.GetSettings(ctx)
+ if err != nil {
+ issues = append(issues, issue(
+ "office_retention_settings_invalid",
+ "Retention settings unreadable",
+ fmt.Sprintf("Stored retention settings could not be read; using the documented defaults. (%s)", err.Error()),
+ ))
+ }
+
+ if _, readable := c.previewMarker.Get(ctx); !readable {
+ issues = append(issues, issue(
+ "office_retention_preview_unreadable",
+ "Retention preview marker unreadable",
+ "The retention preview marker could not be read; office_routine_runs and runs will be previewed again on the next sweep rather than deleting.",
+ ))
+ }
+
+ if lastSweep, ok := c.sweeper.LastSweepSnapshot(); ok {
+ issues = append(issues, sweptTableIssues(lastSweep, settings)...)
+ issues = append(issues, failedTableIssues(lastSweep)...)
+ }
+
+ issues = append(issues, c.censusIssues(settings)...)
+
+ sort.Slice(issues, func(i, j int) bool { return issues[i].ID < issues[j].ID })
+ return issues
+}
+
+// sweptTableIssues covers AC-003.2 (preview pending, one combined issue
+// naming every swept table with a nonzero would-delete count) and AC-003.6
+// (backlog, per swept table).
+func sweptTableIssues(last LastSweep, settings Settings) []health.Issue {
+ var issues []health.Issue
+
+ type sweptEntry struct {
+ table TableName
+ result SweptTableResult
+ window int
+ }
+ entries := []sweptEntry{
+ {TableOfficeRoutineRuns, last.OfficeRoutineRuns, settings.RoutineRuns.WindowDays},
+ {TableRuns, last.Runs, settings.Runs.WindowDays},
+ }
+
+ var pending []string
+ for _, e := range entries {
+ if e.result.Previewed && e.result.WouldDelete > 0 {
+ pending = append(pending, fmt.Sprintf("%s: %d rows under a %d-day window", e.table, e.result.WouldDelete, e.window))
+ }
+ }
+ if len(pending) > 0 {
+ issues = append(issues, issue(
+ "office_retention_preview_pending",
+ "Retention preview pending deletion",
+ "Deletion begins at the next scheduled sweep: "+strings.Join(pending, "; ")+".",
+ ))
+ }
+
+ for _, e := range entries {
+ if !e.result.Backlog {
+ continue
+ }
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_backlog:%s", e.table),
+ "Retention is behind",
+ fmt.Sprintf("%s has more eligible rows than one sweep's batch limit; %d rows were deleted this sweep and retention remains behind.", e.table, e.result.Deleted),
+ ))
+ }
+ return issues
+}
+
+// failedTableIssues covers AC-002.7/AC-004.6's per-table sweep failure. Only
+// office_routine_runs and runs ever carry a nonempty Err in the current
+// sweep implementation — a satellite's own delete is never independently
+// batched or retried — but every reported table is checked generically so a
+// future failure mode on a satellite surfaces without a code change here.
+func failedTableIssues(last LastSweep) []health.Issue {
+ entries := []struct {
+ table TableName
+ result TableSweepResult
+ }{
+ {TableOfficeRoutineRuns, last.OfficeRoutineRuns.TableSweepResult},
+ {TableRuns, last.Runs.TableSweepResult},
+ {TableRunEvents, last.RunEvents},
+ {"office_run_route_attempts", last.RouteAttempts},
+ {"office_run_skills", last.RunSkills},
+ }
+ var issues []health.Issue
+ for _, e := range entries {
+ if e.result.Err == "" {
+ continue
+ }
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_failed:%s", e.table),
+ "Retention sweep failed",
+ fmt.Sprintf("The last sweep failed for %s: %s", e.table, e.result.Err),
+ ))
+ }
+ return issues
+}
+
+// censusIssues covers AC-001.10 (unknown status), AC-003.5/-003.7 (threshold,
+// split on enabled/disabled), and office_retention_count_failed.
+func (c *Checker) censusIssues(settings Settings) []health.Issue {
+ counts := c.sweeper.CensusSnapshot()
+
+ entries := []struct {
+ table TableName
+ census TableCensus
+ warnRows int
+ }{
+ {TableOfficeRoutineRuns, counts.OfficeRoutineRuns, settings.RoutineRuns.WarnRows},
+ {TableRuns, counts.Runs, settings.Runs.WarnRows},
+ {TableRunEvents, counts.RunEvents, settings.RunEvents.WarnRows},
+ }
+
+ var issues []health.Issue
+ for _, e := range entries {
+ if e.census.State == CensusStale {
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_count_failed:%s", e.table),
+ "Retained-row count evaluation failing",
+ fmt.Sprintf("%s's retained-row count could not be re-evaluated; showing the last successful count from %s.", e.table, e.census.AsOf.Format(time.RFC3339)),
+ ))
+ }
+ if e.census.State == CensusNotComputed {
+ continue
+ }
+
+ if len(e.census.UnknownStatuses) > 0 {
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_unknown_status:%s", e.table),
+ "Unrecognized status in retained rows",
+ fmt.Sprintf("%s has rows with unrecognized status values, treated as live state and never pruned: %s.", e.table, formatUnknownStatuses(e.census.UnknownStatuses)),
+ ))
+ }
+
+ if e.warnRows <= 0 || e.census.RetainedCount <= int64(e.warnRows) {
+ continue
+ }
+ message := thresholdMessage(e.table, e.census, e.warnRows)
+ if !settings.Enabled {
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_disabled:%s", e.table),
+ "Retention disabled with rows over threshold",
+ message+" Retention is disabled, so this table is not being pruned.",
+ ))
+ continue
+ }
+ issues = append(issues, issue(
+ fmt.Sprintf("office_retention_threshold:%s", e.table),
+ "Retained rows over threshold",
+ message,
+ ))
+ }
+ return issues
+}
+
+// formatUnknownStatuses renders each unrecognized status with its row
+// count, in the ascending order summarizeStatusCensus already sorted them
+// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10).
+func formatUnknownStatuses(unknown []UnknownStatusCount) string {
+ parts := make([]string, len(unknown))
+ for i, u := range unknown {
+ parts[i] = fmt.Sprintf("%s (%d)", u.Status, u.Count)
+ }
+ return strings.Join(parts, ", ")
+}
+
+func thresholdMessage(table TableName, census TableCensus, warnRows int) string {
+ message := fmt.Sprintf("%s has %d retained rows, over its threshold of %d.", table, census.RetainedCount, warnRows)
+ if census.TopRoutineID != "" {
+ message += fmt.Sprintf(" Routine %s holds the largest share of retained rows, %.1f%%.", census.TopRoutineID, census.TopRoutineShare*100)
+ }
+ return message
+}
+
+func issue(id, title, message string) health.Issue {
+ return health.Issue{
+ ID: id,
+ Category: "office",
+ Title: title,
+ Message: message,
+ Severity: health.SeverityWarning,
+ FixURL: fixURL,
+ FixLabel: fixLabel,
+ }
+}
diff --git a/apps/backend/internal/office/retention/health_test.go b/apps/backend/internal/office/retention/health_test.go
new file mode 100644
index 00000000000..a768495a26b
--- /dev/null
+++ b/apps/backend/internal/office/retention/health_test.go
@@ -0,0 +1,381 @@
+package retention
+
+import (
+ "context"
+ "errors"
+ "strings"
+ "testing"
+
+ "github.com/jmoiron/sqlx"
+
+ "github.com/kandev/kandev/internal/health"
+)
+
+var errCensusEvaluation = errors.New("census evaluation failed")
+
+func newTestChecker(t *testing.T) (*Checker, *Sweeper, *sqlx.DB) {
+ t.Helper()
+ sweeper, conn := newTestSweeper(t)
+ checker := NewChecker(sweeper.settingsStore, sweeper, sweeper.previewMarker)
+ return checker, sweeper, conn
+}
+
+func issueIDs(issues []health.Issue) []string {
+ ids := make([]string, 0, len(issues))
+ for _, i := range issues {
+ ids = append(ids, i.ID)
+ }
+ return ids
+}
+
+func hasIssue(issues []health.Issue, id string) bool {
+ for _, i := range issues {
+ if i.ID == id {
+ return true
+ }
+ }
+ return false
+}
+
+func issueMessage(t *testing.T, issues []health.Issue, id string) string {
+ t.Helper()
+ for _, i := range issues {
+ if i.ID == id {
+ return i.Message
+ }
+ }
+ t.Fatalf("no issue with id %q in %v", id, issueIDs(issues))
+ return ""
+}
+
+func TestChecker_NameAndCategory(t *testing.T) {
+ checker, _, _ := newTestChecker(t)
+ if checker.Name() != "Office run retention" {
+ t.Fatalf("Name() = %q", checker.Name())
+ }
+ if checker.Category() != "office" {
+ t.Fatalf("Category() = %q", checker.Category())
+ }
+}
+
+func TestChecker_FreshInstallNoSweepNoIssues(t *testing.T) {
+ checker, sweeper, _ := newTestChecker(t)
+ ctx := context.Background()
+ sweeper.RunCensus(ctx) // AC-003.11: counts available before any sweep
+
+ issues := checker.Check(ctx)
+ if len(issues) != 0 {
+ t.Fatalf("issues = %v, want none on a fresh, empty, enabled install", issueIDs(issues))
+ }
+}
+
+func TestChecker_SettingsInvalidRaisesGlobalIssue(t *testing.T) {
+ checker, _, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ if _, err := conn.Exec(`
+ INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP)
+ `); err != nil {
+ t.Fatalf("seed unparseable settings: %v", err)
+ }
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_settings_invalid") {
+ t.Fatalf("issues = %v, want office_retention_settings_invalid", issueIDs(issues))
+ }
+}
+
+func TestChecker_PreviewMarkerUnreadableRaisesGlobalIssue(t *testing.T) {
+ checker, _, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ if _, err := conn.Exec(`
+ INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention_preview_completed', 'not json', CURRENT_TIMESTAMP)
+ `); err != nil {
+ t.Fatalf("seed unparseable preview marker: %v", err)
+ }
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_preview_unreadable") {
+ t.Fatalf("issues = %v, want office_retention_preview_unreadable", issueIDs(issues))
+ }
+}
+
+func TestChecker_PreviewPendingNamesTablesWithNonzeroWouldDelete(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ seedRun(t, conn, newID(), "agent-1", "finished", &old, old)
+
+ sweeper.RunSweep(ctx) // preview pass, both tables have 1 eligible row
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_preview_pending") {
+ t.Fatalf("issues = %v, want office_retention_preview_pending", issueIDs(issues))
+ }
+}
+
+func TestChecker_PreviewWithNothingEligibleRaisesNoWarning(t *testing.T) {
+ checker, sweeper, _ := newTestChecker(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ sweeper.RunSweep(ctx) // preview pass, nothing seeded, both tables report zero
+
+ issues := checker.Check(ctx)
+ if hasIssue(issues, "office_retention_preview_pending") {
+ t.Fatalf("issues = %v, want no preview_pending when the preview found nothing (AC-003.3)", issueIDs(issues))
+ }
+}
+
+func TestChecker_BacklogRaisesPerSweptTable(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ const eligibleRows = 105
+ const batchLimit = 100
+
+ seedRoutine(t, conn, "r-1")
+ for i := 0; i < eligibleRows; i++ {
+ old := daysAgo(60 + i)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ }
+ settings := DefaultSettings()
+ settings.BatchLimit = batchLimit
+ settings.RoutineRuns.FloorPerOwner = 0
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ sweeper.RunSweep(ctx) // preview pass
+ sweeper.RunSweep(ctx) // deleting pass, 105 eligible > batch limit 100
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_backlog:office_routine_runs") {
+ t.Fatalf("issues = %v, want office_retention_backlog:office_routine_runs", issueIDs(issues))
+ }
+ if hasIssue(issues, "office_retention_backlog:runs") {
+ t.Fatalf("issues = %v, want no backlog issue for runs (never seeded)", issueIDs(issues))
+ }
+}
+
+func TestChecker_PreviewedTableNeverRaisesBacklog(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ const eligibleRows = 105
+ const batchLimit = 100
+
+ seedRoutine(t, conn, "r-1")
+ for i := 0; i < eligibleRows; i++ {
+ old := daysAgo(60 + i)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ }
+ settings := DefaultSettings()
+ settings.BatchLimit = batchLimit
+ settings.RoutineRuns.FloorPerOwner = 0
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ sweeper.RunSweep(ctx) // preview pass only: 105 eligible, never backlog (AC-002.3)
+
+ issues := checker.Check(ctx)
+ if hasIssue(issues, "office_retention_backlog:office_routine_runs") {
+ t.Fatalf("issues = %v, want no backlog issue for a preview pass regardless of eligible count", issueIDs(issues))
+ }
+}
+
+func TestChecker_AbandonedRunsBatchRaisesFailedIssueForRunsOnly(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ old := daysAgo(60)
+ runID := newID()
+ seedRun(t, conn, runID, "agent-1", "finished", &old, old)
+ seedRunEvent(t, conn, runID, 1)
+
+ sweeper.RunSweep(ctx) // preview pass
+
+ testBeforeSelectEligibleRunIDs = func(attempt int) {
+ if attempt == 0 {
+ return
+ }
+ conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID)
+ }
+ testAfterSelectEligibleRunIDs = func(int, []string) {
+ conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID)
+ }
+ t.Cleanup(func() {
+ testBeforeSelectEligibleRunIDs = nil
+ testAfterSelectEligibleRunIDs = nil
+ })
+
+ sweeper.RunSweep(ctx) // deleting pass: forced to abandon
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_failed:runs") {
+ t.Fatalf("issues = %v, want office_retention_failed:runs", issueIDs(issues))
+ }
+ if hasIssue(issues, "office_retention_failed:run_events") {
+ t.Fatalf("issues = %v, want no failed issue for run_events (the rollback restored it, not a failure of its own)", issueIDs(issues))
+ }
+}
+
+func TestChecker_UnknownStatusRaisesWhileDisabledAndBeforeAnySweep(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.Enabled = false
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2))
+
+ sweeper.RunCensus(ctx) // AC-003.11: census runs independent of the sweep/enabled state
+
+ issues := checker.Check(ctx)
+ const id = "office_retention_unknown_status:office_routine_runs"
+ if !hasIssue(issues, id) {
+ t.Fatalf("issues = %v, want %s (001.10, disabled, no sweep ever ran)", issueIDs(issues), id)
+ }
+ if message := issueMessage(t, issues, id); !strings.Contains(message, "quarantined (2)") {
+ t.Fatalf("message = %q, want it to name the unrecognized status with its row count", message)
+ }
+}
+
+func TestChecker_ThresholdExceededWhileEnabledRaisesThresholdID(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.RoutineRuns.WarnRows = 1
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2))
+
+ sweeper.RunCensus(ctx)
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_threshold:office_routine_runs") {
+ t.Fatalf("issues = %v, want office_retention_threshold:office_routine_runs", issueIDs(issues))
+ }
+ if hasIssue(issues, "office_retention_disabled:office_routine_runs") {
+ t.Fatalf("issues = %v, want no disabled-variant issue while enabled", issueIDs(issues))
+ }
+}
+
+func TestChecker_ThresholdExceededWhileDisabledRaisesDisabledID(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.Enabled = false
+ settings.RoutineRuns.WarnRows = 1
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2))
+
+ sweeper.RunCensus(ctx)
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_disabled:office_routine_runs") {
+ t.Fatalf("issues = %v, want office_retention_disabled:office_routine_runs (AC-003.7)", issueIDs(issues))
+ }
+ if hasIssue(issues, "office_retention_threshold:office_routine_runs") {
+ t.Fatalf("issues = %v, want no plain threshold issue while disabled", issueIDs(issues))
+ }
+}
+
+func TestChecker_ZeroWarnRowsDisablesThresholdIssue(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.RoutineRuns.WarnRows = 0
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ for i := 0; i < 5; i++ {
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(i+1)), daysAgo(i+1))
+ }
+ sweeper.RunCensus(ctx)
+
+ issues := checker.Check(ctx)
+ if hasIssue(issues, "office_retention_threshold:office_routine_runs") || hasIssue(issues, "office_retention_disabled:office_routine_runs") {
+ t.Fatalf("issues = %v, want no threshold issue when warn_rows=0", issueIDs(issues))
+ }
+}
+
+func TestChecker_CountFailedOnlyAfterAPriorSuccess(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ sweeper.RunCensus(ctx) // succeeds once
+
+ sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation)
+
+ issues := checker.Check(ctx)
+ if !hasIssue(issues, "office_retention_count_failed:office_routine_runs") {
+ t.Fatalf("issues = %v, want office_retention_count_failed:office_routine_runs after a prior success then a failure", issueIDs(issues))
+ }
+}
+
+func TestChecker_NotComputedNeverRaisesCountFailed(t *testing.T) {
+ checker, sweeper, _ := newTestChecker(t)
+ ctx := context.Background()
+
+ sweeper.census.RecordRoutineRuns(TableCensus{}, errCensusEvaluation) // fails, no prior success
+
+ issues := checker.Check(ctx)
+ if hasIssue(issues, "office_retention_count_failed:office_routine_runs") {
+ t.Fatalf("issues = %v, want no count_failed issue before any evaluation ever succeeded (AC-003.11's not-yet-computed state)", issueIDs(issues))
+ }
+}
+
+func TestChecker_IssuesSortedByID(t *testing.T) {
+ checker, sweeper, conn := newTestChecker(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.RoutineRuns.WarnRows = 1
+ settings.Runs.WarnRows = 1
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(2)), daysAgo(2))
+ seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1))
+ seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(2)), daysAgo(2))
+ sweeper.RunCensus(ctx)
+
+ issues := checker.Check(ctx)
+ ids := issueIDs(issues)
+ for i := 1; i < len(ids); i++ {
+ if ids[i-1] > ids[i] {
+ t.Fatalf("issues not sorted by id: %v", ids)
+ }
+ }
+}
diff --git a/apps/backend/internal/office/retention/lock.go b/apps/backend/internal/office/retention/lock.go
new file mode 100644
index 00000000000..a96802c3e56
--- /dev/null
+++ b/apps/backend/internal/office/retention/lock.go
@@ -0,0 +1,107 @@
+package retention
+
+import (
+ "context"
+ "database/sql/driver"
+ "time"
+
+ "github.com/jmoiron/sqlx"
+
+ "github.com/kandev/kandev/internal/db"
+)
+
+// advisoryLockKey is retention's own PostgreSQL advisory lock namespace,
+// distinct from every other hashtextextended(?, 0) call site in the repo
+// (participants.go, secrets/sqlite_store.go, workflow/repository/phase2_sqlite.go)
+// so a sweep never contends with participant-seat, secret-transfer, or
+// workflow-phase locking.
+const advisoryLockKey = "office_run_retention_sweep"
+
+const unlockTimeout = 5 * time.Second
+
+// sweepSession is a PostgreSQL session-scoped, non-blocking exclusivity
+// lock for one sweep (AC-OFFICE-RUN-HISTORY-RETENTION-002.12):
+//
+// - Single connection budget: every statement the sweep issues — count,
+// delete, census — runs through this one dedicated connection,
+// returned by queryer(), rather than reserving it for the lock alone
+// and running batches on the shared pool. Reserving a second
+// connection is exactly what a maxOpenConns=1 pool (what
+// testutil.OpenIsolatedPostgres, the mandated Postgres-gated test
+// harness, sets) cannot supply; one connection total removes the
+// deadlock.
+// - Exclusivity window: because every sweep statement runs on the lock
+// connection, there is no window where work proceeds on a different
+// connection after the session died — if the session ends, the very
+// next statement on it fails immediately instead of continuing to run
+// against the pool while another backend has already re-acquired the
+// lock. The between-tables alive() check the design specifies is kept
+// anyway, as a cheap early exit before starting a table's work rather
+// than the only guard against loss of exclusivity.
+// - Release safety: release() unlocks and closes on an independent
+// context, not the sweep's (which may already be cancelled), with a
+// bounded timeout, and discards the connection via driver.ErrBadConn
+// whenever the unlock did not provably succeed — so database/sql
+// never pools a session that may still hold the lock, which is what
+// the design's "cannot wedge retention permanently" claim actually
+// requires (internal/db sets no ConnMaxLifetime).
+type sweepSession struct {
+ conn *sqlx.Conn
+}
+
+// acquireSweepSession tries to take the advisory lock on a fresh dedicated
+// connection. ok is false when another backend already holds it or the
+// connection could not be checked out; the caller records a skip and
+// returns rather than retrying (AC-OFFICE-RUN-HISTORY-RETENTION-002.12).
+func acquireSweepSession(ctx context.Context, pool *db.Pool) (*sweepSession, bool, error) {
+ conn, err := pool.Writer().Connx(ctx)
+ if err != nil {
+ return nil, false, err
+ }
+
+ var acquired bool
+ err = conn.GetContext(ctx, &acquired, `SELECT pg_try_advisory_lock(hashtextextended($1, 0))`, advisoryLockKey)
+ if err != nil {
+ _ = conn.Close()
+ return nil, false, err
+ }
+ if !acquired {
+ _ = conn.Close()
+ return nil, false, nil
+ }
+ return &sweepSession{conn: conn}, true, nil
+}
+
+// queryer is the connection every sweep statement must run through for the
+// whole sweep's duration (see the single-connection-budget and
+// exclusivity-window invariants on sweepSession above).
+func (s *sweepSession) queryer() queryer {
+ return s.conn
+}
+
+// alive reports whether the lock connection is still usable. Checked
+// between tables; the sweep stops before the next table when this returns
+// false and does not attempt to re-acquire, since a re-acquisition after
+// another backend has taken the lock would produce exactly the concurrent
+// sweep AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent.
+func (s *sweepSession) alive(ctx context.Context) bool {
+ return s.conn.PingContext(ctx) == nil
+}
+
+// release unlocks and closes the session. Always safe to call once; never
+// call it twice.
+func (s *sweepSession) release() {
+ ctx, cancel := context.WithTimeout(context.Background(), unlockTimeout)
+ defer cancel()
+
+ var unlocked bool
+ err := s.conn.GetContext(ctx, &unlocked, `SELECT pg_advisory_unlock(hashtextextended($1, 0))`, advisoryLockKey)
+ if err != nil || !unlocked {
+ // The unlock did not provably succeed: force database/sql to
+ // discard this connection instead of returning it to the pool,
+ // so a session that may still hold the lock can never be reused
+ // by a later, unrelated caller.
+ _ = s.conn.Raw(func(driverConn any) error { return driver.ErrBadConn })
+ }
+ _ = s.conn.Close()
+}
diff --git a/apps/backend/internal/office/retention/lock_postgres_test.go b/apps/backend/internal/office/retention/lock_postgres_test.go
new file mode 100644
index 00000000000..8912887e0aa
--- /dev/null
+++ b/apps/backend/internal/office/retention/lock_postgres_test.go
@@ -0,0 +1,141 @@
+package retention
+
+import (
+ "context"
+ "testing"
+
+ "github.com/kandev/kandev/internal/db"
+ "github.com/kandev/kandev/internal/testutil"
+)
+
+// TestSweepSession_SecondBackendSkipsRatherThanBlocks proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-002.12: a second backend racing for the
+// same advisory lock gets ok=false immediately rather than waiting.
+func TestSweepSession_SecondBackendSkipsRatherThanBlocks(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ first := testutil.OpenIsolatedPostgres(t, dsn)
+ firstPool := db.NewPool(first, first)
+ winner, ok, err := acquireSweepSession(ctx, firstPool)
+ if err != nil {
+ t.Fatalf("acquireSweepSession (winner): %v", err)
+ }
+ if !ok {
+ t.Fatal("winner: ok = false, want true")
+ }
+ defer winner.release()
+
+ second := testutil.OpenIsolatedPostgres(t, dsn)
+ secondPool := db.NewPool(second, second)
+ loser, ok, err := acquireSweepSession(ctx, secondPool)
+ if err != nil {
+ t.Fatalf("acquireSweepSession (loser): %v", err)
+ }
+ if ok {
+ loser.release()
+ t.Fatal("loser: ok = true, want false")
+ }
+}
+
+// TestSweepSession_RunsOnMaxOpenConnsOnePool is F27's regression test: the
+// mandated Postgres test harness (testutil.OpenIsolatedPostgres) opens with
+// SetMaxOpenConns(1). Reserving the lock connection AND running batches on
+// the shared pool would deadlock there, since no second connection is ever
+// available. Routing every statement through the lock connection's own
+// queryer() must not deadlock and must give a winner more than one table's
+// worth of work to do, so a transaction-scoped lock would have incorrectly
+// looked sufficient here.
+func TestSweepSession_RunsOnMaxOpenConnsOnePool(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ conn := testutil.OpenIsolatedPostgres(t, dsn)
+ pool := db.NewPool(conn, conn)
+
+ session, ok, err := acquireSweepSession(ctx, pool)
+ if err != nil {
+ t.Fatalf("acquireSweepSession: %v", err)
+ }
+ if !ok {
+ t.Fatal("ok = false, want true")
+ }
+ defer session.release()
+
+ q := session.queryer()
+ for i := 0; i < 3; i++ {
+ var one int
+ if err := q.GetContext(ctx, &one, `SELECT 1`); err != nil {
+ t.Fatalf("query %d on lock connection: %v", i, err)
+ }
+ if one != 1 {
+ t.Fatalf("query %d = %d, want 1", i, one)
+ }
+ }
+}
+
+// TestSweepSession_ReleaseAllowsReacquisition proves release() actually
+// drops the lock rather than merely closing a connection database/sql
+// might still consider live for pooling purposes.
+func TestSweepSession_ReleaseAllowsReacquisition(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+ conn := testutil.OpenIsolatedPostgres(t, dsn)
+ pool := db.NewPool(conn, conn)
+
+ first, ok, err := acquireSweepSession(ctx, pool)
+ if err != nil || !ok {
+ t.Fatalf("first acquire: ok=%v err=%v", ok, err)
+ }
+ first.release()
+
+ second, ok, err := acquireSweepSession(ctx, pool)
+ if err != nil {
+ t.Fatalf("second acquireSweepSession: %v", err)
+ }
+ if !ok {
+ t.Fatal("second acquire: ok = false, want true after release")
+ }
+ second.release()
+}
+
+// TestSweepSession_AliveFalseAfterSessionTerminated proves the
+// between-tables liveness check (F25's early exit) actually detects a
+// dropped session, and that PostgreSQL's own advisory-lock self-healing
+// (F26's justification for a session lock over a lease row) lets a new
+// session acquire afterward.
+func TestSweepSession_AliveFalseAfterSessionTerminated(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ admin := testutil.OpenIsolatedPostgres(t, dsn)
+ adminPool := db.NewPool(admin, admin)
+
+ victimConn := testutil.OpenIsolatedPostgres(t, dsn)
+ victimPool := db.NewPool(victimConn, victimConn)
+ victim, ok, err := acquireSweepSession(ctx, victimPool)
+ if err != nil || !ok {
+ t.Fatalf("victim acquire: ok=%v err=%v", ok, err)
+ }
+
+ var pid int
+ if err := victim.conn.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil {
+ t.Fatalf("select pg_backend_pid: %v", err)
+ }
+ if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil {
+ t.Fatalf("terminate victim backend: %v", err)
+ }
+
+ if victim.alive(ctx) {
+ t.Fatal("alive() = true after backend termination, want false")
+ }
+
+ replacement, ok, err := acquireSweepSession(ctx, adminPool)
+ if err != nil {
+ t.Fatalf("replacement acquireSweepSession: %v", err)
+ }
+ if !ok {
+ t.Fatal("replacement: ok = false, want true (terminated session must release the lock)")
+ }
+ replacement.release()
+}
diff --git a/apps/backend/internal/office/retention/metrics_vars.go b/apps/backend/internal/office/retention/metrics_vars.go
new file mode 100644
index 00000000000..dc12c9cd591
--- /dev/null
+++ b/apps/backend/internal/office/retention/metrics_vars.go
@@ -0,0 +1,55 @@
+package retention
+
+import (
+ "expvar"
+ "strings"
+)
+
+// expvar maps published at package init, exposed via stdlib's /debug/vars
+// handler, mirroring internal/office/scheduler/metrics_vars.go's label
+// model. Development convenience only (see the design's expvar
+// section) — nothing in REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on
+// these; the durable operator surfaces are structured logs and
+// health.Issue.
+var (
+ retentionSweepTotal = expvar.NewMap("office_retention_sweep_total")
+ retentionDeletedTotal = expvar.NewMap("office_retention_deleted_total")
+ retentionCensusTotal = expvar.NewMap("office_retention_census_total")
+)
+
+// metricLabel builds a "k1=v1;k2=v2;..." label string for an expvar map
+// key, matching office/scheduler's format so a downstream parser handles
+// both packages identically.
+func metricLabel(pairs ...string) string {
+ if len(pairs)%2 != 0 {
+ return ""
+ }
+ parts := make([]string, 0, len(pairs)/2)
+ for i := 0; i < len(pairs); i += 2 {
+ parts = append(parts, pairs[i]+"="+pairs[i+1])
+ }
+ return strings.Join(parts, ";")
+}
+
+func incSweepCompleted() {
+ retentionSweepTotal.Add(metricLabel("outcome", "completed"), 1)
+}
+
+func incSweepSkipped() {
+ retentionSweepTotal.Add(metricLabel("outcome", "skipped"), 1)
+}
+
+func incDeleted(table TableName, n int64) {
+ if n <= 0 {
+ return
+ }
+ retentionDeletedTotal.Add(metricLabel("table", string(table)), n)
+}
+
+func incCensus(table TableName, err error) {
+ outcome := "fresh"
+ if err != nil {
+ outcome = "stale"
+ }
+ retentionCensusTotal.Add(metricLabel("table", string(table), "outcome", outcome), 1)
+}
diff --git a/apps/backend/internal/office/retention/metrics_vars_test.go b/apps/backend/internal/office/retention/metrics_vars_test.go
new file mode 100644
index 00000000000..d9b25a1856d
--- /dev/null
+++ b/apps/backend/internal/office/retention/metrics_vars_test.go
@@ -0,0 +1,116 @@
+package retention
+
+import (
+ "errors"
+ "expvar"
+ "strconv"
+ "strings"
+ "testing"
+)
+
+// readCounter walks the expvar map looking for a key that matches the
+// supplied prefix. Returns 0 when no key matches. The prefix match keeps
+// the assertion robust against process-wide test pollution.
+func readCounter(t *testing.T, m *expvar.Map, prefix string) int64 {
+ t.Helper()
+ var total int64
+ m.Do(func(kv expvar.KeyValue) {
+ if !strings.HasPrefix(kv.Key, prefix) {
+ return
+ }
+ n, err := strconv.ParseInt(kv.Value.String(), 10, 64)
+ if err != nil {
+ t.Fatalf("counter %q value not int: %s", kv.Key, kv.Value.String())
+ }
+ total += n
+ })
+ return total
+}
+
+func TestMetricLabel(t *testing.T) {
+ cases := []struct {
+ name string
+ pairs []string
+ want string
+ }{
+ {"single_pair", []string{"table", "runs"}, "table=runs"},
+ {"odd_args_returns_empty", []string{"table"}, ""},
+ {"two_pairs", []string{"table", "runs", "outcome", "completed"}, "table=runs;outcome=completed"},
+ }
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ if got := metricLabel(tc.pairs...); got != tc.want {
+ t.Errorf("metricLabel(%v) = %q, want %q", tc.pairs, got, tc.want)
+ }
+ })
+ }
+}
+
+func TestIncSweepCompletedAndSkipped(t *testing.T) {
+ beforeCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed"))
+ incSweepCompleted()
+ afterCompleted := readCounter(t, retentionSweepTotal, metricLabel("outcome", "completed"))
+ if afterCompleted-beforeCompleted != 1 {
+ t.Errorf("sweep completed delta = %d, want 1", afterCompleted-beforeCompleted)
+ }
+
+ beforeSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped"))
+ incSweepSkipped()
+ afterSkipped := readCounter(t, retentionSweepTotal, metricLabel("outcome", "skipped"))
+ if afterSkipped-beforeSkipped != 1 {
+ t.Errorf("sweep skipped delta = %d, want 1", afterSkipped-beforeSkipped)
+ }
+}
+
+func TestIncDeleted_ZeroIsNotRecorded(t *testing.T) {
+ label := metricLabel("table", "test_table_zero")
+ before := readCounter(t, retentionDeletedTotal, label)
+ incDeleted("test_table_zero", 0)
+ after := readCounter(t, retentionDeletedTotal, label)
+ if after != before {
+ t.Errorf("delta = %d, want 0 (a zero delete must not create a counter entry)", after-before)
+ }
+}
+
+func TestIncDeleted_PositiveIsRecorded(t *testing.T) {
+ label := metricLabel("table", "test_table_positive")
+ before := readCounter(t, retentionDeletedTotal, label)
+ incDeleted("test_table_positive", 7)
+ after := readCounter(t, retentionDeletedTotal, label)
+ if after-before != 7 {
+ t.Errorf("delta = %d, want 7", after-before)
+ }
+}
+
+func TestIncCensus_NilErrorIsFresh(t *testing.T) {
+ label := metricLabel("table", "test_census_fresh", "outcome", "fresh")
+ before := readCounter(t, retentionCensusTotal, label)
+ incCensus("test_census_fresh", nil)
+ after := readCounter(t, retentionCensusTotal, label)
+ if after-before != 1 {
+ t.Errorf("fresh delta = %d, want 1", after-before)
+ }
+}
+
+func TestIncCensus_ErrorIsStale(t *testing.T) {
+ label := metricLabel("table", "test_census_stale", "outcome", "stale")
+ before := readCounter(t, retentionCensusTotal, label)
+ incCensus("test_census_stale", errors.New("boom"))
+ after := readCounter(t, retentionCensusTotal, label)
+ if after-before != 1 {
+ t.Errorf("stale delta = %d, want 1", after-before)
+ }
+}
+
+func TestExpvarMapsPublishedAtKnownNames(t *testing.T) {
+ expected := []string{
+ "office_retention_sweep_total",
+ "office_retention_deleted_total",
+ "office_retention_census_total",
+ }
+ for _, name := range expected {
+ if expvar.Get(name) == nil {
+ t.Errorf("expvar %q not published — /debug/vars consumers will miss it", name)
+ }
+ }
+}
diff --git a/apps/backend/internal/office/retention/policy.go b/apps/backend/internal/office/retention/policy.go
new file mode 100644
index 00000000000..b3b7b4b6370
--- /dev/null
+++ b/apps/backend/internal/office/retention/policy.go
@@ -0,0 +1,59 @@
+package retention
+
+// StatusClass distinguishes a row eligible for age-based deletion from a row
+// a live decision still reads, and from a row whose status this package does
+// not recognize at all — a status classification failure, not a bare bool,
+// so an unrecognized status can be told apart from a recognized live-state
+// one (AC-OFFICE-RUN-HISTORY-RETENTION-001.10).
+type StatusClass int
+
+const (
+ // StatusUnknown is a status belonging to neither a table's history set
+ // nor its live-state set. Treated as live state and warned about.
+ StatusUnknown StatusClass = iota
+ StatusHistory
+ StatusLiveState
+)
+
+// RoutineRunHistoryStatuses are the office_routine_runs statuses eligible
+// for age-based deletion (AC-OFFICE-RUN-HISTORY-RETENTION-001.1).
+var RoutineRunHistoryStatuses = []string{"skipped", "coalesced", "failed", "done", "cancelled"}
+
+// RoutineRunLiveStatuses are the office_routine_runs statuses that are
+// never age-pruned, at any age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1).
+var RoutineRunLiveStatuses = []string{"received", "task_created"}
+
+// RunHistoryStatuses are the runs statuses eligible for age-based deletion
+// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2). cancelled is history: its only
+// writer, CancelRunsWhere, moves a row there from queued/claimed and stamps
+// finished_at in the same statement.
+var RunHistoryStatuses = []string{"finished", "failed", "cancelled"}
+
+// RunLiveStatuses are the runs statuses that are never age-pruned, at any
+// age, including a run parked for a future routing retry
+// (AC-OFFICE-RUN-HISTORY-RETENTION-001.2).
+var RunLiveStatuses = []string{"queued", "claimed"}
+
+// ClassifyRoutineRunStatus classifies an office_routine_runs.status value.
+func ClassifyRoutineRunStatus(status string) StatusClass {
+ return classify(status, RoutineRunHistoryStatuses, RoutineRunLiveStatuses)
+}
+
+// ClassifyRunStatus classifies a runs.status value.
+func ClassifyRunStatus(status string) StatusClass {
+ return classify(status, RunHistoryStatuses, RunLiveStatuses)
+}
+
+func classify(status string, history, live []string) StatusClass {
+ for _, s := range history {
+ if s == status {
+ return StatusHistory
+ }
+ }
+ for _, s := range live {
+ if s == status {
+ return StatusLiveState
+ }
+ }
+ return StatusUnknown
+}
diff --git a/apps/backend/internal/office/retention/policy_test.go b/apps/backend/internal/office/retention/policy_test.go
new file mode 100644
index 00000000000..d4626a1fc20
--- /dev/null
+++ b/apps/backend/internal/office/retention/policy_test.go
@@ -0,0 +1,98 @@
+package retention
+
+import (
+ "sort"
+ "testing"
+
+ "github.com/kandev/kandev/internal/office/models"
+)
+
+func TestClassifyRoutineRunStatus(t *testing.T) {
+ for _, s := range RoutineRunHistoryStatuses {
+ if got := ClassifyRoutineRunStatus(s); got != StatusHistory {
+ t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusHistory", s, got)
+ }
+ }
+ for _, s := range RoutineRunLiveStatuses {
+ if got := ClassifyRoutineRunStatus(s); got != StatusLiveState {
+ t.Errorf("ClassifyRoutineRunStatus(%q) = %v, want StatusLiveState", s, got)
+ }
+ }
+ if got := ClassifyRoutineRunStatus("some_future_status"); got != StatusUnknown {
+ t.Errorf("ClassifyRoutineRunStatus(unrecognized) = %v, want StatusUnknown", got)
+ }
+}
+
+func TestClassifyRunStatus(t *testing.T) {
+ for _, s := range RunHistoryStatuses {
+ if got := ClassifyRunStatus(s); got != StatusHistory {
+ t.Errorf("ClassifyRunStatus(%q) = %v, want StatusHistory", s, got)
+ }
+ }
+ for _, s := range RunLiveStatuses {
+ if got := ClassifyRunStatus(s); got != StatusLiveState {
+ t.Errorf("ClassifyRunStatus(%q) = %v, want StatusLiveState", s, got)
+ }
+ }
+ if got := ClassifyRunStatus("some_future_status"); got != StatusUnknown {
+ t.Errorf("ClassifyRunStatus(unrecognized) = %v, want StatusUnknown", got)
+ }
+}
+
+// TestRoutineRunStatusSets_CoverEveryEnumValue pins the closed-set claim in
+// the system design: history + live-state must equal exactly the
+// office/models RoutineRunStatus enumeration. A future eighth status added
+// to enums.go without updating this file fails here rather than silently
+// falling through ClassifyRoutineRunStatus's StatusUnknown branch on every
+// install that never runs this test — the exact gap that hid `cancelled`.
+func TestRoutineRunStatusSets_CoverEveryEnumValue(t *testing.T) {
+ wantAll := []string{
+ models.RoutineRunStatusReceived.String(),
+ models.RoutineRunStatusTaskCreated.String(),
+ models.RoutineRunStatusSkipped.String(),
+ models.RoutineRunStatusCoalesced.String(),
+ models.RoutineRunStatusFailed.String(),
+ models.RoutineRunStatusDone.String(),
+ models.RoutineRunStatusCancelled.String(),
+ }
+ gotAll := append(append([]string{}, RoutineRunHistoryStatuses...), RoutineRunLiveStatuses...)
+ assertSameSet(t, "office_routine_runs", wantAll, gotAll)
+}
+
+// TestRunStatusSets_CoverEveryLiteralSQLWriter pins the closed-set claim for
+// runs.status. models.RunStatus itself is incomplete (it does not list
+// "cancelled", even though CancelRunsWhere in
+// internal/runs/repository/sqlite/cancel.go writes it) — this test asserts
+// against the actual literal statuses written by SQL in that package
+// instead, per the system design's Testing section, so the enum's own
+// incompleteness cannot mask a gap here the way it masked `cancelled`
+// before this capability existed.
+func TestRunStatusSets_CoverEveryLiteralSQLWriter(t *testing.T) {
+ // Literal statuses written by internal/runs/repository/sqlite:
+ // cancel.go: 'cancelled' (CancelRunsWhere)
+ // runs.go:172: 'claimed' (ClaimNextRun family)
+ // runs.go:512: 'claimed' (claim by reason)
+ // runs.go:539: 'queued' (ScheduleRetry)
+ // runs.go:562: 'queued' (recoverStaleClaimed)
+ // runs.go:195/737: status passed as a bound parameter, produced by
+ // callers with 'finished' or 'failed' (FinishRun/MarkRunFailed).
+ wantAll := []string{"queued", "claimed", "finished", "failed", "cancelled"}
+ gotAll := append(append([]string{}, RunHistoryStatuses...), RunLiveStatuses...)
+ assertSameSet(t, "runs", wantAll, gotAll)
+}
+
+func assertSameSet(t *testing.T, table string, want, got []string) {
+ t.Helper()
+ w := append([]string{}, want...)
+ g := append([]string{}, got...)
+ sort.Strings(w)
+ sort.Strings(g)
+ if len(w) != len(g) {
+ t.Fatalf("%s: status set size = %d, want %d (got=%v want=%v)", table, len(g), len(w), g, w)
+ }
+ for i := range w {
+ if w[i] != g[i] {
+ t.Fatalf("%s: status set mismatch at %d: got %v, want %v", table, i, g, w)
+ }
+ }
+}
diff --git a/apps/backend/internal/office/retention/preview_marker.go b/apps/backend/internal/office/retention/preview_marker.go
new file mode 100644
index 00000000000..ef8942298f4
--- /dev/null
+++ b/apps/backend/internal/office/retention/preview_marker.go
@@ -0,0 +1,119 @@
+package retention
+
+import (
+ "context"
+ "database/sql"
+ "encoding/json"
+ "errors"
+ "time"
+
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+// previewMarkerKey is the settings-store key holding the per-swept-table
+// preview completion marker, a JSON object keyed by table name.
+const previewMarkerKey = "office_run_retention_preview_completed"
+
+// PreviewMarker records, per swept table, the timestamp its preview
+// completed. A table absent from the map has never completed a preview.
+type PreviewMarker map[TableName]time.Time
+
+// PreviewMarkerStore persists and reads the preview marker.
+type PreviewMarkerStore struct {
+ store *systemsettings.Store
+}
+
+// NewPreviewMarkerStore wraps the shared key/value settings store.
+func NewPreviewMarkerStore(store *systemsettings.Store) *PreviewMarkerStore {
+ return &PreviewMarkerStore{store: store}
+}
+
+// Get reads the marker. readable is false only when a document is present
+// but cannot be read or parsed (AC-OFFICE-RUN-HISTORY-RETENTION-003.10); a
+// document that was never written is the ordinary fresh-install state and
+// reports readable=true with an empty marker, not an error. Either way, a
+// table absent from the returned marker has not completed a preview.
+func (s *PreviewMarkerStore) Get(ctx context.Context) (PreviewMarker, bool) {
+ raw, found, err := s.store.Get(ctx, previewMarkerKey)
+ if err != nil {
+ return PreviewMarker{}, false
+ }
+ if !found {
+ return PreviewMarker{}, true
+ }
+ var doc PreviewMarker
+ if err := json.Unmarshal(raw, &doc); err != nil {
+ return PreviewMarker{}, false
+ }
+ if doc == nil {
+ doc = PreviewMarker{}
+ }
+ return doc, true
+}
+
+// MarkCompleted records that table's preview as completed at the given
+// time. It is only ever called after that table's preview evaluation
+// completed successfully, is never cleared by a settings change or
+// restart, and never touches another table's entry. If the stored document
+// was corrupt, this write replaces it with a fresh document carrying only
+// this table's entry — the safe direction, since a spurious re-preview of
+// another table costs one sweep and deletes nothing.
+func (s *PreviewMarkerStore) MarkCompleted(ctx context.Context, table TableName, at time.Time) error {
+ marker, readable := s.Get(ctx)
+ if !readable {
+ marker = PreviewMarker{}
+ }
+ marker[table] = at
+ raw, err := json.Marshal(marker)
+ if err != nil {
+ return err
+ }
+ return s.store.Save(ctx, previewMarkerKey, raw)
+}
+
+// GetWith is Get against an explicit connection instead of the shared
+// settings pool. Required mid-sweep on PostgreSQL: every statement in a
+// sweep must run on the session holding the advisory lock (see sweep.go
+// and lock.go), and going through the pool here would request a second
+// connection, which deadlocks under a maxOpenConns=1 pool (the mandated
+// Postgres-gated test harness). This bypasses systemsettings.Store and
+// reads the key/value pair directly, so it depends on that package's
+// `settings` table keeping its `key`/`value` column names.
+func (s *PreviewMarkerStore) GetWith(ctx context.Context, q queryer) (PreviewMarker, bool) {
+ var raw string
+ err := q.GetContext(ctx, &raw, q.Rebind(`SELECT value FROM settings WHERE key = ?`), previewMarkerKey)
+ if err != nil {
+ if errors.Is(err, sql.ErrNoRows) {
+ return PreviewMarker{}, true
+ }
+ return PreviewMarker{}, false
+ }
+ var doc PreviewMarker
+ if err := json.Unmarshal([]byte(raw), &doc); err != nil {
+ return PreviewMarker{}, false
+ }
+ if doc == nil {
+ doc = PreviewMarker{}
+ }
+ return doc, true
+}
+
+// MarkCompletedWith is MarkCompleted against an explicit connection; see
+// GetWith.
+func (s *PreviewMarkerStore) MarkCompletedWith(ctx context.Context, q queryer, table TableName, at time.Time) error {
+ marker, readable := s.GetWith(ctx, q)
+ if !readable {
+ marker = PreviewMarker{}
+ }
+ marker[table] = at
+ raw, err := json.Marshal(marker)
+ if err != nil {
+ return err
+ }
+ _, err = q.ExecContext(ctx, q.Rebind(`
+ INSERT INTO settings (key, value, updated_at)
+ VALUES (?, ?, ?)
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at
+ `), previewMarkerKey, string(raw), time.Now().UTC())
+ return err
+}
diff --git a/apps/backend/internal/office/retention/preview_marker_test.go b/apps/backend/internal/office/retention/preview_marker_test.go
new file mode 100644
index 00000000000..8080eb380dc
--- /dev/null
+++ b/apps/backend/internal/office/retention/preview_marker_test.go
@@ -0,0 +1,117 @@
+package retention
+
+import (
+ "context"
+ "testing"
+ "time"
+)
+
+func TestPreviewMarkerStore_MissingIsReadableAndEmpty(t *testing.T) {
+ _, raw := newTestSettingsStore(t)
+ store := NewPreviewMarkerStore(raw)
+
+ marker, readable := store.Get(context.Background())
+ if !readable {
+ t.Fatalf("Get() readable = false, want true for a never-written marker")
+ }
+ if len(marker) != 0 {
+ t.Fatalf("Get() marker = %+v, want empty", marker)
+ }
+}
+
+func TestPreviewMarkerStore_MarkCompletedThenGetRoundTrips(t *testing.T) {
+ _, raw := newTestSettingsStore(t)
+ store := NewPreviewMarkerStore(raw)
+ ctx := context.Background()
+
+ at := time.Date(2026, 9, 9, 12, 0, 0, 0, time.UTC)
+ if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, at); err != nil {
+ t.Fatalf("MarkCompleted: %v", err)
+ }
+
+ marker, readable := store.Get(ctx)
+ if !readable {
+ t.Fatalf("Get() readable = false after a valid write")
+ }
+ got, ok := marker[TableOfficeRoutineRuns]
+ if !ok {
+ t.Fatalf("marker missing office_routine_runs entry: %+v", marker)
+ }
+ if !got.Equal(at) {
+ t.Fatalf("marker[office_routine_runs] = %v, want %v", got, at)
+ }
+ if _, ok := marker[TableRuns]; ok {
+ t.Fatalf("marker has an entry for runs before it was ever marked: %+v", marker)
+ }
+}
+
+// TestPreviewMarkerStore_PerTableNotGlobal proves the marker is per swept
+// table, not per database: marking one table previewed must not mark a
+// sibling table previewed too (AC-OFFICE-RUN-HISTORY-RETENTION-003.4).
+func TestPreviewMarkerStore_PerTableNotGlobal(t *testing.T) {
+ _, raw := newTestSettingsStore(t)
+ store := NewPreviewMarkerStore(raw)
+ ctx := context.Background()
+
+ if err := store.MarkCompleted(ctx, TableOfficeRoutineRuns, time.Now().UTC()); err != nil {
+ t.Fatalf("MarkCompleted(office_routine_runs): %v", err)
+ }
+
+ marker, _ := store.Get(ctx)
+ if _, ok := marker[TableRuns]; ok {
+ t.Fatalf("marking office_routine_runs previewed also marked runs: %+v", marker)
+ }
+
+ if err := store.MarkCompleted(ctx, TableRuns, time.Now().UTC()); err != nil {
+ t.Fatalf("MarkCompleted(runs): %v", err)
+ }
+ marker, _ = store.Get(ctx)
+ if len(marker) != 2 {
+ t.Fatalf("marker after both tables previewed = %+v, want 2 entries", marker)
+ }
+}
+
+// TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-003.10: a present-but-corrupt marker
+// document is treated as "not yet previewed" for every swept table, which
+// is the safe direction because a spurious re-preview deletes nothing.
+func TestPreviewMarkerStore_UnparseableTreatsEveryTableAsNotPreviewed(t *testing.T) {
+ _, raw := newTestSettingsStore(t)
+ ctx := context.Background()
+ if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil {
+ t.Fatalf("seed unparseable marker: %v", err)
+ }
+
+ store := NewPreviewMarkerStore(raw)
+ marker, readable := store.Get(ctx)
+ if readable {
+ t.Fatalf("Get() readable = true for an unparseable marker, want false")
+ }
+ if len(marker) != 0 {
+ t.Fatalf("Get() marker = %+v on unparseable document, want empty (not-previewed)", marker)
+ }
+}
+
+// TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable proves that
+// marking a table previewed after a corrupt document is detected replaces
+// the corrupt document rather than erroring forever.
+func TestPreviewMarkerStore_MarkCompletedRecoversFromUnparseable(t *testing.T) {
+ _, raw := newTestSettingsStore(t)
+ ctx := context.Background()
+ if err := raw.Save(ctx, previewMarkerKey, []byte("not json")); err != nil {
+ t.Fatalf("seed unparseable marker: %v", err)
+ }
+ store := NewPreviewMarkerStore(raw)
+
+ at := time.Now().UTC()
+ if err := store.MarkCompleted(ctx, TableRuns, at); err != nil {
+ t.Fatalf("MarkCompleted after corrupt document: %v", err)
+ }
+ marker, readable := store.Get(ctx)
+ if !readable {
+ t.Fatalf("Get() readable = false after a fresh valid write")
+ }
+ if _, ok := marker[TableRuns]; !ok {
+ t.Fatalf("marker missing runs entry after recovery write: %+v", marker)
+ }
+}
diff --git a/apps/backend/internal/office/retention/runtime.go b/apps/backend/internal/office/retention/runtime.go
new file mode 100644
index 00000000000..06af274b44a
--- /dev/null
+++ b/apps/backend/internal/office/retention/runtime.go
@@ -0,0 +1,61 @@
+package retention
+
+import (
+ "context"
+
+ "github.com/kandev/kandev/internal/db"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+// Runtime composes every retention component behind one Start/Stop pair,
+// mirroring internal/system/storage.Runtime's shape: a scheduler goroutine,
+// an HTTP handler, and a health checker, wired together once at boot over
+// the shared pool and the shared key/value settings store.
+type Runtime struct {
+ Store *Store
+ SettingsStore *SettingsStore
+ PreviewMarker *PreviewMarkerStore
+ Sweeper *Sweeper
+ Scheduler *Scheduler
+ Checker *Checker
+ Handler *Handler
+}
+
+// NewRuntime constructs every retention component. logError, when set, is
+// forwarded to Handler for a failed settings read or write; it never
+// affects the sweep, which fails closed on its own terms (see
+// SettingsStore.GetSettingsForSweep).
+func NewRuntime(pool *db.Pool, settingsStore *systemsettings.Store, logError func(string, error)) *Runtime {
+ store := NewStore(pool)
+ retentionSettings := NewSettingsStore(settingsStore)
+ previewMarker := NewPreviewMarkerStore(settingsStore)
+ sweeper := NewSweeper(pool, store, retentionSettings, previewMarker)
+ scheduler := NewScheduler(retentionSettings, sweeper, SchedulerOptions{})
+ checker := NewChecker(retentionSettings, sweeper, previewMarker)
+ handler := NewHandler(HandlerConfig{
+ SettingsStore: retentionSettings,
+ Sweeper: sweeper,
+ OnSettingsChanged: scheduler.ApplySettings,
+ LogError: logError,
+ })
+ return &Runtime{
+ Store: store,
+ SettingsStore: retentionSettings,
+ PreviewMarker: previewMarker,
+ Sweeper: sweeper,
+ Scheduler: scheduler,
+ Checker: checker,
+ Handler: handler,
+ }
+}
+
+// Start begins the scheduler loop (census immediately, sweep gated by
+// enablement — see scheduler.go). Idempotent: safe to call once at boot.
+func (r *Runtime) Start(ctx context.Context) error {
+ return r.Scheduler.Start(ctx)
+}
+
+// Stop cancels and joins the scheduler loop.
+func (r *Runtime) Stop() {
+ r.Scheduler.Stop()
+}
diff --git a/apps/backend/internal/office/retention/runtime_test.go b/apps/backend/internal/office/retention/runtime_test.go
new file mode 100644
index 00000000000..684097e4a5c
--- /dev/null
+++ b/apps/backend/internal/office/retention/runtime_test.go
@@ -0,0 +1,79 @@
+package retention
+
+import (
+ "testing"
+
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/db"
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+func newTestRuntime(t *testing.T) *Runtime {
+ t.Helper()
+ conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init office schema: %v", err)
+ }
+ pool := db.NewPool(conn, conn)
+ settingsStore, err := systemsettings.NewStore(pool)
+ if err != nil {
+ t.Fatalf("init settings schema: %v", err)
+ }
+ return NewRuntime(pool, settingsStore, nil)
+}
+
+func TestNewRuntime_WiresEveryComponentNonNil(t *testing.T) {
+ runtime := newTestRuntime(t)
+ if runtime.Store == nil || runtime.SettingsStore == nil || runtime.PreviewMarker == nil ||
+ runtime.Sweeper == nil || runtime.Scheduler == nil || runtime.Checker == nil || runtime.Handler == nil {
+ t.Fatalf("Runtime has a nil component: %+v", runtime)
+ }
+}
+
+func TestRuntime_StartStopIsClean(t *testing.T) {
+ runtime := newTestRuntime(t)
+ ctx := t.Context()
+
+ if err := runtime.Start(ctx); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer runtime.Stop()
+
+ if err := runtime.Start(ctx); err != nil {
+ t.Fatalf("second Start: %v", err)
+ }
+ runtime.Stop()
+ runtime.Stop() // idempotent
+}
+
+func TestRuntime_HandlerOnSettingsChangedReachesScheduler(t *testing.T) {
+ runtime := newTestRuntime(t)
+ ctx := t.Context()
+ if err := runtime.Start(ctx); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer runtime.Stop()
+
+ settings := DefaultSettings()
+ settings.SweepIntervalHours = 2
+ saved, err := runtime.SettingsStore.SaveSettings(ctx, settings)
+ if err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ // Handler's OnSettingsChanged is wired to Scheduler.ApplySettings; call
+ // it exactly as putRetention would and confirm the running scheduler
+ // picked up the change (AC-004.5, without a restart).
+ runtime.Handler.config.OnSettingsChanged(saved)
+ if got := runtime.Scheduler.latestSettings(); got.SweepIntervalHours != 2 {
+ t.Fatalf("Scheduler.latestSettings().SweepIntervalHours = %d, want 2", got.SweepIntervalHours)
+ }
+}
diff --git a/apps/backend/internal/office/retention/scheduler.go b/apps/backend/internal/office/retention/scheduler.go
new file mode 100644
index 00000000000..ec0725783eb
--- /dev/null
+++ b/apps/backend/internal/office/retention/scheduler.go
@@ -0,0 +1,192 @@
+package retention
+
+import (
+ "context"
+ "sync"
+ "time"
+)
+
+// firstSweepDelay is the fixed delay before the first sweep after Start,
+// and after retention transitions from disabled to enabled
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.10, -002.13). Arming at the full
+// interval instead — as the census timer does — would mean an install
+// restarted more often than the interval never sweeps at all.
+const firstSweepDelay = 5 * time.Minute
+
+// SchedulerOptions configures Scheduler construction. After is injectable
+// for deterministic tests; production leaves it nil and gets time.After.
+type SchedulerOptions struct {
+ After func(time.Duration) <-chan time.Time
+}
+
+// Scheduler owns two independent timers on one goroutine, modelled on
+// internal/system/storage.Scheduler:
+//
+// - The census timer always runs, on the configured sweep interval,
+// whether or not retention is enabled (AC-OFFICE-RUN-HISTORY-RETENTION-003.11):
+// a disabled install is exactly the one whose tables grow unattended,
+// and the only one where an unrecognized status would otherwise never
+// be noticed.
+// - The sweep timer only runs while enabled, armed at firstSweepDelay
+// after Start or after a disabled-to-enabled transition, and at the
+// full interval thereafter (AC-OFFICE-RUN-HISTORY-RETENTION-002.10,
+// -002.13).
+//
+// A settings change wakes the loop and re-arms both timers from the
+// moment of the change (fixed-delay, not fixed-rate): the sweep timer at
+// firstSweepDelay only when retention just turned on, otherwise at the
+// (possibly new) full interval; the census timer always at the full
+// interval. The census timer also refreshes the shared settings record, so a
+// backend that did not serve a settings write still adopts it while disabled.
+type Scheduler struct {
+ settingsStore *SettingsStore
+ sweeper *Sweeper
+ after func(time.Duration) <-chan time.Time
+
+ lifecycleMu sync.Mutex
+ mu sync.Mutex
+ cancel context.CancelFunc
+ wake chan struct{}
+ latest Settings
+ wg sync.WaitGroup
+}
+
+// NewScheduler wires the scheduler to its dependencies.
+func NewScheduler(settingsStore *SettingsStore, sweeper *Sweeper, options SchedulerOptions) *Scheduler {
+ after := options.After
+ if after == nil {
+ after = time.After
+ }
+ return &Scheduler{settingsStore: settingsStore, sweeper: sweeper, after: after}
+}
+
+// Start begins the scheduler loop. A no-op when already running.
+func (s *Scheduler) Start(ctx context.Context) error {
+ s.lifecycleMu.Lock()
+ defer s.lifecycleMu.Unlock()
+ s.mu.Lock()
+ running := s.cancel != nil
+ s.mu.Unlock()
+ if running {
+ return nil
+ }
+
+ // GetSettings never fails outright: an unreadable or unparseable stored
+ // document still yields DefaultSettings, so the scheduler starts on
+ // those defaults rather than never starting at all. The wrapped error
+ // is reported separately through Checker.Check.
+ settings, _ := s.settingsStore.GetSettings(ctx)
+
+ s.mu.Lock()
+ workerCtx, cancel := context.WithCancel(ctx)
+ s.cancel = cancel
+ s.wake = make(chan struct{}, 1)
+ s.latest = settings
+ s.wg.Add(1)
+ wake := s.wake
+ s.mu.Unlock()
+
+ go s.run(workerCtx, settings, wake)
+ return nil
+}
+
+// ApplySettings notifies a running scheduler that settings changed,
+// re-arming both timers from this moment. A no-op when not running.
+func (s *Scheduler) ApplySettings(settings Settings) {
+ s.mu.Lock()
+ if s.cancel == nil || s.wake == nil {
+ s.mu.Unlock()
+ return
+ }
+ s.latest = settings
+ wake := s.wake
+ s.mu.Unlock()
+ select {
+ case wake <- struct{}{}:
+ default:
+ }
+}
+
+// Stop cancels the scheduler loop and joins it. A no-op when not running.
+func (s *Scheduler) Stop() {
+ s.lifecycleMu.Lock()
+ defer s.lifecycleMu.Unlock()
+ s.mu.Lock()
+ cancel := s.cancel
+ s.cancel = nil
+ s.wake = nil
+ s.mu.Unlock()
+ if cancel != nil {
+ cancel()
+ s.wg.Wait()
+ }
+}
+
+func (s *Scheduler) run(ctx context.Context, settings Settings, wake <-chan struct{}) {
+ defer s.wg.Done()
+
+ // The first census evaluation runs here, on the scheduler goroutine,
+ // not blocking Start's caller (AC-OFFICE-RUN-HISTORY-RETENTION-003.11's
+ // "runs at Start, not at the first sweep").
+ s.sweeper.RunCensus(ctx)
+ census := s.after(sweepInterval(settings))
+
+ var sweep <-chan time.Time
+ if settings.Enabled {
+ sweep = s.after(firstSweepDelay)
+ }
+
+ for {
+ select {
+ case <-ctx.Done():
+ return
+ case <-wake:
+ wasEnabled := settings.Enabled
+ settings = s.latestSettings()
+
+ sweep = nil
+ if settings.Enabled {
+ if wasEnabled {
+ sweep = s.after(sweepInterval(settings))
+ } else {
+ sweep = s.after(firstSweepDelay)
+ }
+ }
+ census = s.after(sweepInterval(settings))
+
+ case <-census:
+ s.sweeper.RunCensus(ctx)
+ if latest, err := s.settingsStore.GetSettings(ctx); err == nil && latest != settings {
+ wasEnabled := settings.Enabled
+ settings = latest
+ s.mu.Lock()
+ s.latest = latest
+ s.mu.Unlock()
+
+ sweep = nil
+ if settings.Enabled {
+ if wasEnabled {
+ sweep = s.after(sweepInterval(settings))
+ } else {
+ sweep = s.after(firstSweepDelay)
+ }
+ }
+ }
+ census = s.after(sweepInterval(settings))
+
+ case <-sweep:
+ s.sweeper.RunSweep(ctx)
+ sweep = s.after(sweepInterval(settings))
+ }
+ }
+}
+
+func (s *Scheduler) latestSettings() Settings {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ return s.latest
+}
+
+func sweepInterval(settings Settings) time.Duration {
+ return time.Duration(settings.SweepIntervalHours) * time.Hour
+}
diff --git a/apps/backend/internal/office/retention/scheduler_test.go b/apps/backend/internal/office/retention/scheduler_test.go
new file mode 100644
index 00000000000..b790497c5a9
--- /dev/null
+++ b/apps/backend/internal/office/retention/scheduler_test.go
@@ -0,0 +1,351 @@
+package retention
+
+import (
+ "context"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/db"
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+// fakeAfter is a deterministic stand-in for time.After, keyed by duration:
+// every call for the same duration returns the same channel, so a test can
+// fire a specific timer (e.g. firstSweepDelay vs. the configured interval)
+// without racing real wall-clock time. armed records every duration the
+// scheduler has requested, in order, so a test can prove which timer was
+// armed and when — including proving RunCensus already completed
+// synchronously before the following after() call.
+type fakeAfter struct {
+ mu sync.Mutex
+ chans map[time.Duration]chan time.Time
+ armed chan time.Duration
+}
+
+func newFakeAfter() *fakeAfter {
+ return &fakeAfter{chans: map[time.Duration]chan time.Time{}, armed: make(chan time.Duration, 64)}
+}
+
+func (f *fakeAfter) after(d time.Duration) <-chan time.Time {
+ f.mu.Lock()
+ ch, ok := f.chans[d]
+ if !ok {
+ ch = make(chan time.Time, 1)
+ f.chans[d] = ch
+ }
+ f.mu.Unlock()
+ f.armed <- d
+ return ch
+}
+
+func (f *fakeAfter) fire(t *testing.T, d time.Duration) {
+ t.Helper()
+ f.mu.Lock()
+ ch, ok := f.chans[d]
+ f.mu.Unlock()
+ if !ok {
+ t.Fatalf("fire: no timer ever armed for duration %v", d)
+ }
+ ch <- time.Now()
+}
+
+// waitArmed blocks until after() has been called with duration d,
+// draining (and discarding) any other durations seen along the way.
+func (f *fakeAfter) waitArmed(t *testing.T, d time.Duration) {
+ t.Helper()
+ deadline := time.After(2 * time.Second)
+ for {
+ select {
+ case got := <-f.armed:
+ if got == d {
+ return
+ }
+ case <-deadline:
+ t.Fatalf("timed out waiting for a timer to be armed at %v", d)
+ }
+ }
+}
+
+// assertNotArmed drains any pending arm notifications and fails if d is
+// among them.
+func (f *fakeAfter) assertNotArmed(t *testing.T, d time.Duration) {
+ t.Helper()
+ for {
+ select {
+ case got := <-f.armed:
+ if got == d {
+ t.Fatalf("timer armed at %v, want it never armed", d)
+ }
+ default:
+ return
+ }
+ }
+}
+
+func newTestScheduler(t *testing.T, settings Settings, fake *fakeAfter) (*Scheduler, *Sweeper) {
+ t.Helper()
+ sweeper, _ := newTestSweeper(t)
+ if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+ scheduler := NewScheduler(sweeper.settingsStore, sweeper, SchedulerOptions{After: fake.after})
+ return scheduler, sweeper
+}
+
+func TestScheduler_RunsCensusAtStartRegardlessOfEnabled(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = false
+ scheduler, sweeper := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+
+ // The next after() call is census's re-arm, issued only once RunCensus
+ // (called synchronously beforehand in run()) has returned.
+ fake.waitArmed(t, sweepInterval(settings))
+
+ counts := sweeper.CensusSnapshot()
+ if counts.OfficeRoutineRuns.State != CensusFresh {
+ t.Fatalf("office_routine_runs census state = %v, want fresh (disabled must not block the census)", counts.OfficeRoutineRuns.State)
+ }
+ if counts.Runs.State != CensusFresh || counts.RunEvents.State != CensusFresh {
+ t.Fatalf("census not fresh for every table: %+v", counts)
+ }
+}
+
+func TestScheduler_SweepArmedAtFirstDelayWhenEnabledAtStart(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = true
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+
+ fake.waitArmed(t, firstSweepDelay)
+}
+
+func TestScheduler_SweepNotArmedWhenDisabledAtStart(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = false
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+
+ fake.waitArmed(t, sweepInterval(settings)) // census's arm proves the loop is running
+ fake.assertNotArmed(t, firstSweepDelay)
+}
+
+func TestScheduler_SweepFiresAndReArmsAtFullIntervalNotFirstDelayAgain(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = true
+ settings.SweepIntervalHours = 1
+ scheduler, sweeper := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+
+ fake.waitArmed(t, firstSweepDelay)
+ fake.fire(t, firstSweepDelay)
+
+ fake.waitArmed(t, sweepInterval(settings)) // re-armed at the full interval, not another 5-minute delay
+
+ if _, ok := sweeper.LastSweepSnapshot(); !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true (the fired timer must have run a sweep)")
+ }
+}
+
+func TestScheduler_EnablingFromDisabledArmsAtFirstDelay(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = false
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+ fake.waitArmed(t, sweepInterval(settings))
+
+ enabled := settings
+ enabled.Enabled = true
+ scheduler.ApplySettings(enabled)
+
+ fake.waitArmed(t, firstSweepDelay)
+}
+
+func TestScheduler_SettingsChangeWhileEnabledReArmsAtNewIntervalNotFirstDelay(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = true
+ settings.SweepIntervalHours = 1
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+ fake.waitArmed(t, firstSweepDelay) // initial arm; timer never fired
+
+ changed := settings
+ changed.SweepIntervalHours = 2
+ scheduler.ApplySettings(changed)
+
+ fake.waitArmed(t, sweepInterval(changed)) // re-armed at the new interval, not firstSweepDelay again
+}
+
+func TestScheduler_DisablingStopsArmingSweepButNotCensus(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = true
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+ fake.waitArmed(t, firstSweepDelay)
+
+ disabled := settings
+ disabled.Enabled = false
+ scheduler.ApplySettings(disabled)
+
+ fake.waitArmed(t, sweepInterval(disabled)) // census's re-arm proves the wake was processed
+ fake.assertNotArmed(t, firstSweepDelay)
+}
+
+func TestScheduler_ReconcilesSharedSettingsOnCensus(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = false
+ settings.SweepIntervalHours = 1
+ scheduler, sweeper := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+ fake.waitArmed(t, sweepInterval(settings))
+
+ updated := settings
+ updated.Enabled = true
+ if _, err := sweeper.settingsStore.SaveSettings(context.Background(), updated); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+ fake.fire(t, sweepInterval(settings))
+
+ // The census timer is also the periodic shared-settings reconciliation
+ // point. Enabling retention in another backend must arm this process's
+ // first sweep even when no local PUT delivered ApplySettings.
+ fake.waitArmed(t, firstSweepDelay)
+}
+
+func TestScheduler_StartTwiceIsNoop(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = false
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("first Start: %v", err)
+ }
+ defer scheduler.Stop()
+ fake.waitArmed(t, sweepInterval(settings))
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("second Start: %v", err)
+ }
+ // A second run() goroutine would double-arm; draining once more must
+ // time out rather than find another immediate arm.
+ select {
+ case d := <-fake.armed:
+ t.Fatalf("second Start armed another timer at %v", d)
+ case <-time.After(100 * time.Millisecond):
+ }
+}
+
+// TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: an unreadable stored settings
+// document must not disable retention silently. GetSettings already falls
+// back to DefaultSettings on such a document (see settings_store_test.go);
+// this proves Start actually uses that fallback and runs the loop instead
+// of aborting before the goroutine ever spawns.
+func TestScheduler_StartsOnDefaultsWhenStoredSettingsUnparseable(t *testing.T) {
+ fake := newFakeAfter()
+ conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init office schema: %v", err)
+ }
+ pool := db.NewPool(conn, conn)
+ settingsRaw, err := systemsettings.NewStore(pool)
+ if err != nil {
+ t.Fatalf("init settings schema: %v", err)
+ }
+ if err := settingsRaw.Save(context.Background(), settingsKey, []byte("not json")); err != nil {
+ t.Fatalf("seed unparseable settings: %v", err)
+ }
+
+ settingsStore := NewSettingsStore(settingsRaw)
+ sweeper := NewSweeper(pool, NewStore(pool), settingsStore, NewPreviewMarkerStore(settingsRaw))
+ scheduler := NewScheduler(settingsStore, sweeper, SchedulerOptions{After: fake.after})
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ defer scheduler.Stop()
+
+ // DefaultSettings().Enabled is true, so both timers must arm off the
+ // fallback defaults rather than the loop never starting at all.
+ fake.waitArmed(t, sweepInterval(DefaultSettings()))
+ fake.waitArmed(t, firstSweepDelay)
+
+ counts := sweeper.CensusSnapshot()
+ if counts.OfficeRoutineRuns.State != CensusFresh {
+ t.Fatalf("office_routine_runs census state = %v, want fresh (Start must run the census off defaults)", counts.OfficeRoutineRuns.State)
+ }
+}
+
+func TestScheduler_StopJoinsWithoutHanging(t *testing.T) {
+ fake := newFakeAfter()
+ settings := DefaultSettings()
+ settings.Enabled = true
+ scheduler, _ := newTestScheduler(t, settings, fake)
+
+ if err := scheduler.Start(context.Background()); err != nil {
+ t.Fatalf("Start: %v", err)
+ }
+ fake.waitArmed(t, firstSweepDelay)
+
+ done := make(chan struct{})
+ go func() {
+ scheduler.Stop()
+ close(done)
+ }()
+ select {
+ case <-done:
+ case <-time.After(2 * time.Second):
+ t.Fatal("Stop did not return")
+ }
+}
diff --git a/apps/backend/internal/office/retention/settings_store.go b/apps/backend/internal/office/retention/settings_store.go
new file mode 100644
index 00000000000..d7cf6d53cb5
--- /dev/null
+++ b/apps/backend/internal/office/retention/settings_store.go
@@ -0,0 +1,91 @@
+package retention
+
+import (
+ "context"
+ "encoding/json"
+ "fmt"
+
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+// settingsKey is the internal/system/settings.Store key holding the
+// retention policy document. The live key/value table is "settings",
+// reached through settings.Store; system_settings is a legacy SQLite-only
+// table read once for migration and is absent on PostgreSQL.
+const settingsKey = "office_run_retention"
+
+// SettingsStore persists and reads the retention policy document.
+type SettingsStore struct {
+ store *systemsettings.Store
+}
+
+// NewSettingsStore wraps the shared key/value settings store.
+func NewSettingsStore(store *systemsettings.Store) *SettingsStore {
+ return &SettingsStore{store: store}
+}
+
+// GetSettings reads the retention policy for reporting and startup use. An
+// unreadable or unparseable document falls back to DefaultSettings and
+// wraps ErrInvalidPersistedSettings so the caller can raise a health issue;
+// it never fails outright (AC-OFFICE-RUN-HISTORY-RETENTION-004.4).
+func (s *SettingsStore) GetSettings(ctx context.Context) (Settings, error) {
+ raw, found, err := s.store.Get(ctx, settingsKey)
+ if err != nil {
+ return DefaultSettings(), fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err)
+ }
+ if !found {
+ return DefaultSettings(), nil
+ }
+ return decodeAndNormalize(raw, DefaultSettings())
+}
+
+// GetSettingsForSweep reads the retention policy on the writer pool at
+// sweep start, per AC-OFFICE-RUN-HISTORY-RETENTION-004.5: a sweep must read
+// the stored settings at its start rather than a value cached from a
+// notification, so a backend that did not serve the write still sweeps
+// under the new policy. Unlike GetSettings, a read or parse failure here
+// returns ErrInvalidPersistedSettings with no usable Settings value — the
+// caller must skip the sweep rather than fall back to the (possibly
+// shorter) default window and delete history the operator configured the
+// system to keep. A document that was never saved is the legitimate empty
+// state, not a failure, and yields the defaults.
+func (s *SettingsStore) GetSettingsForSweep(ctx context.Context) (Settings, error) {
+ raw, found, err := s.store.GetConsistent(ctx, settingsKey)
+ if err != nil {
+ return Settings{}, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err)
+ }
+ if !found {
+ return DefaultSettings(), nil
+ }
+ return decodeAndNormalize(raw, Settings{})
+}
+
+func decodeAndNormalize(raw []byte, fallback Settings) (Settings, error) {
+ var doc Settings
+ if err := json.Unmarshal(raw, &doc); err != nil {
+ return fallback, fmt.Errorf("%w: decode JSON: %w", ErrInvalidPersistedSettings, err)
+ }
+ normalized, err := NormalizeSettings(doc)
+ if err != nil {
+ return fallback, fmt.Errorf("%w: %w", ErrInvalidPersistedSettings, err)
+ }
+ return normalized, nil
+}
+
+// SaveSettings normalizes and persists a full settings document, replacing
+// whatever was stored (AC-OFFICE-RUN-HISTORY-RETENTION-004.9). A rejected
+// write leaves the stored document unchanged.
+func (s *SettingsStore) SaveSettings(ctx context.Context, in Settings) (Settings, error) {
+ normalized, err := NormalizeSettings(in)
+ if err != nil {
+ return Settings{}, err
+ }
+ raw, err := json.Marshal(normalized)
+ if err != nil {
+ return Settings{}, fmt.Errorf("encode retention settings: %w", err)
+ }
+ if err := s.store.Save(ctx, settingsKey, raw); err != nil {
+ return Settings{}, err
+ }
+ return normalized, nil
+}
diff --git a/apps/backend/internal/office/retention/settings_store_test.go b/apps/backend/internal/office/retention/settings_store_test.go
new file mode 100644
index 00000000000..eacfb89e789
--- /dev/null
+++ b/apps/backend/internal/office/retention/settings_store_test.go
@@ -0,0 +1,154 @@
+package retention
+
+import (
+ "context"
+ "errors"
+ "testing"
+
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/db"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+func newTestSettingsStore(t *testing.T) (*SettingsStore, *systemsettings.Store) {
+ t.Helper()
+ conn, err := sqlx.Open("sqlite3", ":memory:")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ raw, err := systemsettings.NewStore(db.NewPool(conn, conn))
+ if err != nil {
+ t.Fatalf("new settings store: %v", err)
+ }
+ return NewSettingsStore(raw), raw
+}
+
+func TestSettingsStore_MissingReturnsDefaults(t *testing.T) {
+ store, _ := newTestSettingsStore(t)
+ got, err := store.GetSettings(context.Background())
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if got != DefaultSettings() {
+ t.Fatalf("GetSettings() = %+v, want defaults", got)
+ }
+}
+
+func TestSettingsStore_SaveThenGetRoundTrips(t *testing.T) {
+ store, _ := newTestSettingsStore(t)
+ ctx := context.Background()
+ in := DefaultSettings()
+ in.SweepIntervalHours = 12
+ in.RoutineRuns.WindowDays = 90
+
+ saved, err := store.SaveSettings(ctx, in)
+ if err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+ if saved != in {
+ t.Fatalf("SaveSettings returned %+v, want %+v", saved, in)
+ }
+
+ got, err := store.GetSettings(ctx)
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if got != in {
+ t.Fatalf("GetSettings() = %+v, want %+v", got, in)
+ }
+}
+
+func TestSettingsStore_SaveRejectsOutOfRangeAndLeavesStoredUnchanged(t *testing.T) {
+ store, _ := newTestSettingsStore(t)
+ ctx := context.Background()
+ in := DefaultSettings()
+ in.SweepIntervalHours = 12
+ if _, err := store.SaveSettings(ctx, in); err != nil {
+ t.Fatalf("seed SaveSettings: %v", err)
+ }
+
+ bad := DefaultSettings()
+ bad.BatchLimit = 1
+ if _, err := store.SaveSettings(ctx, bad); !errors.Is(err, ErrValidation) {
+ t.Fatalf("SaveSettings(bad): err = %v, want ErrValidation", err)
+ }
+
+ got, err := store.GetSettings(ctx)
+ if err != nil {
+ t.Fatalf("GetSettings: %v", err)
+ }
+ if got.SweepIntervalHours != 12 {
+ t.Fatalf("stored settings changed after rejected write: %+v", got)
+ }
+}
+
+// TestSettingsStore_UnparseableFallsBackToDefaultsForReporting proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-004.4: reading for reporting/startup
+// tolerates an unreadable document and yields the documented defaults
+// (wrapped in ErrInvalidPersistedSettings so a caller can raise a health
+// issue), never failing outright.
+func TestSettingsStore_UnparseableFallsBackToDefaultsForReporting(t *testing.T) {
+ store, raw := newTestSettingsStore(t)
+ ctx := context.Background()
+ if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil {
+ t.Fatalf("seed unparseable settings: %v", err)
+ }
+
+ got, err := store.GetSettings(ctx)
+ if !errors.Is(err, ErrInvalidPersistedSettings) {
+ t.Fatalf("GetSettings: err = %v, want ErrInvalidPersistedSettings", err)
+ }
+ if got != DefaultSettings() {
+ t.Fatalf("GetSettings() = %+v, want defaults on unparseable document", got)
+ }
+}
+
+// TestSettingsStore_ForSweepFailsClosedOnUnparseable proves the other half
+// of AC-OFFICE-RUN-HISTORY-RETENTION-004.5: reading settings *to delete by*
+// must not silently fall back to the (possibly shorter) default window. The
+// caller sees ErrInvalidPersistedSettings and a zero Settings value, and is
+// expected to skip the sweep rather than run it under defaults.
+func TestSettingsStore_ForSweepFailsClosedOnUnparseable(t *testing.T) {
+ store, raw := newTestSettingsStore(t)
+ ctx := context.Background()
+ if err := raw.Save(ctx, settingsKey, []byte("not json")); err != nil {
+ t.Fatalf("seed unparseable settings: %v", err)
+ }
+
+ _, err := store.GetSettingsForSweep(ctx)
+ if !errors.Is(err, ErrInvalidPersistedSettings) {
+ t.Fatalf("GetSettingsForSweep: err = %v, want ErrInvalidPersistedSettings", err)
+ }
+}
+
+func TestSettingsStore_ForSweepMissingUsesDefaults(t *testing.T) {
+ store, _ := newTestSettingsStore(t)
+ got, err := store.GetSettingsForSweep(context.Background())
+ if err != nil {
+ t.Fatalf("GetSettingsForSweep: %v", err)
+ }
+ if got != DefaultSettings() {
+ t.Fatalf("GetSettingsForSweep() = %+v, want defaults when nothing stored", got)
+ }
+}
+
+func TestSettingsStore_ForSweepReadsWriterPool(t *testing.T) {
+ store, _ := newTestSettingsStore(t)
+ ctx := context.Background()
+ in := DefaultSettings()
+ in.RoutineRuns.WindowDays = 3650
+ if _, err := store.SaveSettings(ctx, in); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+ got, err := store.GetSettingsForSweep(ctx)
+ if err != nil {
+ t.Fatalf("GetSettingsForSweep: %v", err)
+ }
+ if got.RoutineRuns.WindowDays != 3650 {
+ t.Fatalf("GetSettingsForSweep() = %+v, want the just-saved window", got)
+ }
+}
diff --git a/apps/backend/internal/office/retention/settings_wire.go b/apps/backend/internal/office/retention/settings_wire.go
new file mode 100644
index 00000000000..73af2563374
--- /dev/null
+++ b/apps/backend/internal/office/retention/settings_wire.go
@@ -0,0 +1,181 @@
+package retention
+
+import (
+ "bytes"
+ "encoding/json"
+ "fmt"
+ "io"
+)
+
+// retentionSettingsEnvelope captures each top-level field as raw JSON
+// rather than a typed value, so a request body can be told apart into
+// three cases per field: absent (nil RawMessage, take the documented
+// default), present as the literal JSON null (rejected —
+// AC-OFFICE-RUN-HISTORY-RETENTION-004.9), or present with a value (decoded
+// and validated). A plain typed struct cannot distinguish the first two: an
+// ordinary `*int` field is nil either way.
+type retentionSettingsEnvelope struct {
+ Enabled json.RawMessage `json:"enabled"`
+ SweepIntervalHours json.RawMessage `json:"sweep_interval_hours"`
+ BatchLimit json.RawMessage `json:"batch_limit"`
+ RoutineRuns json.RawMessage `json:"routine_runs"`
+ Runs json.RawMessage `json:"runs"`
+ RunEvents json.RawMessage `json:"run_events"`
+}
+
+type tableSettingsEnvelope struct {
+ WindowDays json.RawMessage `json:"window_days"`
+ FloorPerOwner json.RawMessage `json:"floor_per_owner"`
+ WarnRows json.RawMessage `json:"warn_rows"`
+}
+
+type runEventsSettingsEnvelope struct {
+ WarnRows json.RawMessage `json:"warn_rows"`
+}
+
+// decodeRetentionSettings implements AC-OFFICE-RUN-HISTORY-RETENTION-004.9's
+// PUT semantics: a full replace where an omitted field takes its documented
+// default, an unrecognized field or an explicit null is rejected naming the
+// field, and nothing is written on any rejection (the caller is expected to
+// not persist the zero-value Settings returned alongside a non-nil error).
+// Range validation (AC-004.3) is NormalizeSettings's job, called by
+// SettingsStore.SaveSettings after this decode succeeds.
+func decodeRetentionSettings(body []byte) (Settings, error) {
+ defaults := DefaultSettings()
+ trimmed := bytes.TrimSpace(body)
+ if len(trimmed) == 0 || trimmed[0] != '{' {
+ return Settings{}, fmt.Errorf("request body: expected a JSON object")
+ }
+
+ dec := json.NewDecoder(bytes.NewReader(trimmed))
+ dec.DisallowUnknownFields()
+ var env retentionSettingsEnvelope
+ if err := dec.Decode(&env); err != nil {
+ return Settings{}, err
+ }
+ var extra any
+ if err := dec.Decode(&extra); err != io.EOF {
+ if err == nil {
+ return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object")
+ }
+ return Settings{}, fmt.Errorf("request body: unexpected data after the JSON object: %w", err)
+ }
+
+ enabled, err := decodeBoolField(env.Enabled, "enabled", defaults.Enabled)
+ if err != nil {
+ return Settings{}, err
+ }
+ sweepIntervalHours, err := decodeIntField(env.SweepIntervalHours, "sweep_interval_hours", defaults.SweepIntervalHours)
+ if err != nil {
+ return Settings{}, err
+ }
+ batchLimit, err := decodeIntField(env.BatchLimit, "batch_limit", defaults.BatchLimit)
+ if err != nil {
+ return Settings{}, err
+ }
+ routineRuns, err := decodeTableSettings(env.RoutineRuns, "routine_runs", defaults.RoutineRuns)
+ if err != nil {
+ return Settings{}, err
+ }
+ runs, err := decodeTableSettings(env.Runs, "runs", defaults.Runs)
+ if err != nil {
+ return Settings{}, err
+ }
+ runEvents, err := decodeRunEventsSettings(env.RunEvents, "run_events", defaults.RunEvents)
+ if err != nil {
+ return Settings{}, err
+ }
+
+ return Settings{
+ Enabled: enabled,
+ SweepIntervalHours: sweepIntervalHours,
+ BatchLimit: batchLimit,
+ RoutineRuns: routineRuns,
+ Runs: runs,
+ RunEvents: runEvents,
+ }, nil
+}
+
+func decodeTableSettings(raw json.RawMessage, name string, defaults TableSettings) (TableSettings, error) {
+ if raw == nil {
+ return defaults, nil
+ }
+ if isJSONNull(raw) {
+ return TableSettings{}, rejectNull(name)
+ }
+ dec := json.NewDecoder(bytes.NewReader(raw))
+ dec.DisallowUnknownFields()
+ var env tableSettingsEnvelope
+ if err := dec.Decode(&env); err != nil {
+ return TableSettings{}, fmt.Errorf("%s: %w", name, err)
+ }
+ windowDays, err := decodeIntField(env.WindowDays, name+".window_days", defaults.WindowDays)
+ if err != nil {
+ return TableSettings{}, err
+ }
+ floorPerOwner, err := decodeIntField(env.FloorPerOwner, name+".floor_per_owner", defaults.FloorPerOwner)
+ if err != nil {
+ return TableSettings{}, err
+ }
+ warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows)
+ if err != nil {
+ return TableSettings{}, err
+ }
+ return TableSettings{WindowDays: windowDays, FloorPerOwner: floorPerOwner, WarnRows: warnRows}, nil
+}
+
+func decodeRunEventsSettings(raw json.RawMessage, name string, defaults RunEventsSettings) (RunEventsSettings, error) {
+ if raw == nil {
+ return defaults, nil
+ }
+ if isJSONNull(raw) {
+ return RunEventsSettings{}, rejectNull(name)
+ }
+ dec := json.NewDecoder(bytes.NewReader(raw))
+ dec.DisallowUnknownFields()
+ var env runEventsSettingsEnvelope
+ if err := dec.Decode(&env); err != nil {
+ return RunEventsSettings{}, fmt.Errorf("%s: %w", name, err)
+ }
+ warnRows, err := decodeIntField(env.WarnRows, name+".warn_rows", defaults.WarnRows)
+ if err != nil {
+ return RunEventsSettings{}, err
+ }
+ return RunEventsSettings{WarnRows: warnRows}, nil
+}
+
+func decodeBoolField(raw json.RawMessage, name string, defaultVal bool) (bool, error) {
+ if raw == nil {
+ return defaultVal, nil
+ }
+ if isJSONNull(raw) {
+ return false, rejectNull(name)
+ }
+ var v bool
+ if err := json.Unmarshal(raw, &v); err != nil {
+ return false, fmt.Errorf("%s: %w", name, err)
+ }
+ return v, nil
+}
+
+func decodeIntField(raw json.RawMessage, name string, defaultVal int) (int, error) {
+ if raw == nil {
+ return defaultVal, nil
+ }
+ if isJSONNull(raw) {
+ return 0, rejectNull(name)
+ }
+ var v int
+ if err := json.Unmarshal(raw, &v); err != nil {
+ return 0, fmt.Errorf("%s: %w", name, err)
+ }
+ return v, nil
+}
+
+func isJSONNull(raw json.RawMessage) bool {
+ return bytes.Equal(bytes.TrimSpace(raw), []byte("null"))
+}
+
+func rejectNull(name string) error {
+ return fmt.Errorf("%s: must not be null; omit the field to use its default", name)
+}
diff --git a/apps/backend/internal/office/retention/store.go b/apps/backend/internal/office/retention/store.go
new file mode 100644
index 00000000000..52d80b8b138
--- /dev/null
+++ b/apps/backend/internal/office/retention/store.go
@@ -0,0 +1,476 @@
+package retention
+
+import (
+ "context"
+ "database/sql"
+ "errors"
+ "fmt"
+ "time"
+
+ "github.com/jmoiron/sqlx"
+
+ "github.com/kandev/kandev/internal/db"
+ "github.com/kandev/kandev/internal/db/dialect"
+)
+
+// queryer is the subset of *sqlx.DB / *sqlx.Conn this package needs to run
+// a sweep or a census. The sweep runs every statement for its whole
+// duration through one queryer: the writer pool directly on SQLite, or one
+// dedicated connection on PostgreSQL (see lock.go) — so the advisory lock
+// and the batches it protects can never diverge onto different sessions.
+type queryer interface {
+ db.Rebinder
+ ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error)
+ QueryxContext(ctx context.Context, query string, args ...any) (*sqlx.Rows, error)
+ QueryRowxContext(ctx context.Context, query string, args ...any) *sqlx.Row
+ GetContext(ctx context.Context, dest any, query string, args ...any) error
+ SelectContext(ctx context.Context, dest any, query string, args ...any) error
+ BeginTxx(ctx context.Context, opts *sql.TxOptions) (*sqlx.Tx, error)
+}
+
+// Store is the raw SQL access layer for retention: the eligibility counts
+// (used by both the preview and the backlog check), the batch deletes, and
+// the status census. It holds no state of its own.
+type Store struct {
+ pool *db.Pool
+}
+
+// NewStore wraps the database pool office_routine_runs, runs, and their
+// satellites live in.
+func NewStore(pool *db.Pool) *Store {
+ return &Store{pool: pool}
+}
+
+// IsPostgres reports whether the writer pool is PostgreSQL. On SQLite one
+// backend process owns the database file, so the in-process sweeping guard
+// is sufficient and no advisory lock is taken.
+func (s *Store) IsPostgres() bool {
+ return dialect.IsPostgres(s.pool.Writer().DriverName())
+}
+
+func routineRunEligibleSubquery() string {
+ return `
+ SELECT id,
+ COALESCE(completed_at, created_at) AS completion_time,
+ ROW_NUMBER() OVER (
+ PARTITION BY routine_id
+ ORDER BY COALESCE(completed_at, created_at) DESC, id DESC
+ ) AS rn
+ FROM office_routine_runs
+ WHERE status IN (?)`
+}
+
+func runEligibleSubquery() string {
+ return `
+ SELECT id,
+ COALESCE(finished_at, requested_at) AS completion_time,
+ ROW_NUMBER() OVER (
+ PARTITION BY agent_profile_id
+ ORDER BY COALESCE(finished_at, requested_at) DESC, id DESC
+ ) AS rn
+ FROM runs
+ WHERE status IN (?)
+ AND NOT EXISTS (
+ SELECT 1
+ FROM office_agent_pause_recoveries
+ WHERE office_agent_pause_recoveries.failed_run_id = runs.id
+ )`
+}
+
+// CountEligibleRoutineRuns is the office_routine_runs eligibility count,
+// uncapped by any batch limit: used for both the preview's WouldDelete
+// (AC-OFFICE-RUN-HISTORY-RETENTION-003.9) and the backlog determination
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3) so the two can never disagree
+// about what "eligible" means.
+func (s *Store) CountEligibleRoutineRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) {
+ query := `SELECT COUNT(*) FROM (` + routineRunEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?`
+ return countEligible(ctx, q, query, RoutineRunHistoryStatuses, floor, cutoff)
+}
+
+// CountEligibleRuns is the runs table's equivalent of CountEligibleRoutineRuns.
+func (s *Store) CountEligibleRuns(ctx context.Context, q queryer, cutoff time.Time, floor int) (int64, error) {
+ query := `SELECT COUNT(*) FROM (` + runEligibleSubquery() + `) ranked WHERE rn > ? AND completion_time < ?`
+ return countEligible(ctx, q, query, RunHistoryStatuses, floor, cutoff)
+}
+
+func countEligible(ctx context.Context, q queryer, query string, statuses []string, floor int, cutoff time.Time) (int64, error) {
+ bound, args, err := db.Bind(q, query, statuses, floor, cutoff)
+ if err != nil {
+ return 0, err
+ }
+ var count int64
+ if err := q.QueryRowxContext(ctx, bound, args...).Scan(&count); err != nil {
+ return 0, err
+ }
+ return count, nil
+}
+
+// DeleteRoutineRunsBatch deletes at most batchLimit eligible
+// office_routine_runs rows, oldest first, re-asserting status, age and the
+// per-owner floor in the same statement
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.3, -002.4). One statement is
+// already atomic, satisfying -002.5 without an explicit transaction.
+func (s *Store) DeleteRoutineRunsBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (int64, error) {
+ query := `
+ DELETE FROM office_routine_runs
+ WHERE id IN (
+ SELECT id FROM (` + routineRunEligibleSubquery() + `) ranked
+ WHERE rn > ? AND completion_time < ?
+ ORDER BY completion_time ASC, id ASC
+ LIMIT ?
+ )
+ AND status IN (?)
+ AND COALESCE(completed_at, created_at) < ?`
+ bound, args, err := db.Bind(q, query,
+ RoutineRunHistoryStatuses, floor, cutoff, batchLimit,
+ RoutineRunHistoryStatuses, cutoff,
+ )
+ if err != nil {
+ return 0, err
+ }
+ res, err := q.ExecContext(ctx, bound, args...)
+ if err != nil {
+ return 0, err
+ }
+ return res.RowsAffected()
+}
+
+// RunBatchResult reports what one runs batch (and its satellites) actually
+// deleted. Abandoned is true only when the batch was rolled back twice in a
+// row and gave up — every count is then zero, matching what the rollback
+// left committed (AC-OFFICE-RUN-HISTORY-RETENTION-002.7).
+type RunBatchResult struct {
+ RunsDeleted int64
+ RunEventsDeleted int64
+ RouteAttemptsDeleted int64
+ RunSkillsDeleted int64
+ Abandoned bool
+}
+
+// DeleteRunBatch selects up to batchLimit eligible runs, deletes each
+// selected run's satellites and then the run itself in one transaction,
+// re-asserting the whole eligibility predicate (status, age, and floor) at
+// delete time. If a concurrent ScheduleRetry resurrects a row between
+// selection and delete, step 5's affected-row count falls short of the
+// selected id count; the whole transaction is rolled back and retried once
+// with a fresh selection. A second mismatch abandons the batch
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4, -002.7).
+func (s *Store) DeleteRunBatch(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) (RunBatchResult, error) {
+ for attempt := 0; attempt < 2; attempt++ {
+ if testBeforeSelectEligibleRunIDs != nil {
+ testBeforeSelectEligibleRunIDs(attempt)
+ }
+ ids, err := s.selectEligibleRunIDs(ctx, q, cutoff, floor, batchLimit)
+ if err != nil {
+ return RunBatchResult{}, err
+ }
+ if len(ids) == 0 {
+ return RunBatchResult{}, nil
+ }
+ if testAfterSelectEligibleRunIDs != nil {
+ testAfterSelectEligibleRunIDs(attempt, ids)
+ }
+ result, matched, err := s.deleteRunBatchOnce(ctx, q, ids, cutoff, floor)
+ if err != nil {
+ return RunBatchResult{}, err
+ }
+ if matched {
+ return result, nil
+ }
+ }
+ return RunBatchResult{Abandoned: true}, nil
+}
+
+// testBeforeSelectEligibleRunIDs and testAfterSelectEligibleRunIDs, when
+// set, bracket each of DeleteRunBatch's (at most two) selections — a
+// deterministic seam this package's own tests use to exercise the
+// selection-to-delete resurrection race, including the two-consecutive-
+// mismatches abandon path, without depending on cross-connection goroutine
+// timing. Never set outside tests.
+var (
+ testBeforeSelectEligibleRunIDs func(attempt int)
+ testAfterSelectEligibleRunIDs func(attempt int, ids []string)
+)
+
+func (s *Store) selectEligibleRunIDs(ctx context.Context, q queryer, cutoff time.Time, floor, batchLimit int) ([]string, error) {
+ query := `
+ SELECT id FROM (` + runEligibleSubquery() + `) ranked
+ WHERE rn > ? AND completion_time < ?
+ ORDER BY completion_time ASC, id ASC
+ LIMIT ?`
+ bound, args, err := db.Bind(q, query, RunHistoryStatuses, floor, cutoff, batchLimit)
+ if err != nil {
+ return nil, err
+ }
+ rows, err := q.QueryxContext(ctx, bound, args...)
+ if err != nil {
+ return nil, err
+ }
+ defer func() { _ = rows.Close() }()
+ var ids []string
+ for rows.Next() {
+ var id string
+ if err := rows.Scan(&id); err != nil {
+ return nil, err
+ }
+ ids = append(ids, id)
+ }
+ return ids, rows.Err()
+}
+
+func (s *Store) deleteRunBatchOnce(ctx context.Context, q queryer, ids []string, cutoff time.Time, floor int) (RunBatchResult, bool, error) {
+ tx, err := q.BeginTxx(ctx, nil)
+ if err != nil {
+ return RunBatchResult{}, false, err
+ }
+ committed := false
+ defer func() {
+ if !committed {
+ _ = tx.Rollback()
+ }
+ }()
+
+ runEventsDeleted, err := deleteByRunIDs(ctx, tx, "run_events", ids)
+ if err != nil {
+ return RunBatchResult{}, false, err
+ }
+ routeAttemptsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_route_attempts", ids)
+ if err != nil {
+ return RunBatchResult{}, false, err
+ }
+ runSkillsDeleted, err := deleteByRunIDs(ctx, tx, "office_run_skills", ids)
+ if err != nil {
+ return RunBatchResult{}, false, err
+ }
+
+ runsDeleted, err := deleteRunsByIDs(ctx, tx, ids, cutoff, floor)
+ if err != nil {
+ return RunBatchResult{}, false, err
+ }
+ if runsDeleted != int64(len(ids)) {
+ return RunBatchResult{}, false, nil
+ }
+
+ if err := tx.Commit(); err != nil {
+ return RunBatchResult{}, false, err
+ }
+ committed = true
+ return RunBatchResult{
+ RunsDeleted: runsDeleted,
+ RunEventsDeleted: runEventsDeleted,
+ RouteAttemptsDeleted: routeAttemptsDeleted,
+ RunSkillsDeleted: runSkillsDeleted,
+ }, true, nil
+}
+
+// retentionMaxHostParams caps id-list placeholders per statement.
+// batch_limit's documented range (AC-OFFICE-RUN-HISTORY-RETENTION-004.3)
+// permits up to 100,000, which would otherwise bind that many ids in one IN
+// clause and can overflow SQLite's compiled variable-count limit or
+// PostgreSQL's wire-protocol parameter cap. Matches the bound this repo
+// already uses for the same reason (internal/task/repository/sqlite's
+// sqliteMaxHostParams).
+const retentionMaxHostParams = 500
+
+// chunkIDs splits ids into sub-slices of at most size entries so an
+// IN-clause query built from them stays under retentionMaxHostParams
+// regardless of batch_limit. An empty input returns nil rather than one
+// empty chunk, since an empty IN () clause is a SQL syntax error.
+func chunkIDs(ids []string, size int) [][]string {
+ if len(ids) == 0 {
+ return nil
+ }
+ if size <= 0 || len(ids) <= size {
+ return [][]string{ids}
+ }
+ chunks := make([][]string, 0, (len(ids)+size-1)/size)
+ for i := 0; i < len(ids); i += size {
+ end := i + size
+ if end > len(ids) {
+ end = len(ids)
+ }
+ chunks = append(chunks, ids[i:end])
+ }
+ return chunks
+}
+
+// deleteRunsByIDs deletes ids from runs, in chunks of at most
+// retentionMaxHostParams. Each chunk's outer WHERE re-asserts status and age
+// directly, in addition to the floor-checking subquery: PostgreSQL's
+// EvalPlanQual recheck of a concurrently updated row re-evaluates a direct
+// column predicate against the row's fresh values, but does not rebuild an
+// uncorrelated id-membership subquery, so the subquery alone is not enough
+// to exclude a row resurrected between selection and delete
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.4).
+func deleteRunsByIDs(ctx context.Context, tx *sqlx.Tx, ids []string, cutoff time.Time, floor int) (int64, error) {
+ var total int64
+ for _, chunk := range chunkIDs(ids, retentionMaxHostParams) {
+ query := `
+ DELETE FROM runs
+ WHERE id IN (?)
+ AND id IN (
+ SELECT id FROM (` + runEligibleSubquery() + `) ranked
+ WHERE rn > ? AND completion_time < ?
+ )
+ AND status IN (?)
+ AND COALESCE(finished_at, requested_at) < ?`
+ bound, args, err := db.Bind(tx, query, chunk, RunHistoryStatuses, floor, cutoff, RunHistoryStatuses, cutoff)
+ if err != nil {
+ return 0, err
+ }
+ res, err := tx.ExecContext(ctx, bound, args...)
+ if err != nil {
+ return 0, err
+ }
+ affected, err := res.RowsAffected()
+ if err != nil {
+ return 0, err
+ }
+ total += affected
+ }
+ return total, nil
+}
+
+// deleteByRunIDs deletes run-id-keyed satellite rows for one table. table
+// must be a hardcoded identifier from a call site in this package, never a
+// caller-supplied string — it is interpolated directly into the query text.
+func deleteByRunIDs(ctx context.Context, tx *sqlx.Tx, table string, ids []string) (int64, error) {
+ var total int64
+ for _, chunk := range chunkIDs(ids, retentionMaxHostParams) {
+ query := fmt.Sprintf(`DELETE FROM %s WHERE run_id IN (?)`, table)
+ bound, args, err := db.Bind(tx, query, chunk)
+ if err != nil {
+ return 0, err
+ }
+ res, err := tx.ExecContext(ctx, bound, args...)
+ if err != nil {
+ return 0, err
+ }
+ affected, err := res.RowsAffected()
+ if err != nil {
+ return 0, err
+ }
+ total += affected
+ }
+ return total, nil
+}
+
+// CountRunEvents is run_events' plain retained count: it has no status
+// column, so it keeps a bare COUNT(*) rather than a census.
+func (s *Store) CountRunEvents(ctx context.Context, q queryer) (int64, error) {
+ var count int64
+ if err := q.GetContext(ctx, &count, `SELECT COUNT(*) FROM run_events`); err != nil {
+ return 0, err
+ }
+ return count, nil
+}
+
+// CensusRoutineRuns issues office_routine_runs' status census
+// (AC-OFFICE-RUN-HISTORY-RETENTION-001.10, -003.5, -003.11). The
+// unknown-status detector needs every distinct status value, which the
+// top-routine attribution's aggregation does not carry, so it stays a
+// separate GROUP BY status scan. The retained total and the top-routine
+// attribution, by contrast, must agree with each other by construction —
+// AC-003.5's share is defined as one routine's retained rows as a
+// proportion of the table's retained count — so routineRunCensusTotals
+// reads both from one statement rather than two, which a concurrent write
+// between them could otherwise make disagree.
+func (s *Store) CensusRoutineRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) {
+ statusCounts, err := statusCensus(ctx, q, "office_routine_runs")
+ if err != nil {
+ return TableCensus{}, err
+ }
+ _, unknown := summarizeStatusCensus(statusCounts, RoutineRunHistoryStatuses, RoutineRunLiveStatuses)
+
+ if testBetweenRoutineRunCensusReads != nil {
+ testBetweenRoutineRunCensusReads(q)
+ }
+
+ retained, topID, topCount, err := routineRunCensusTotals(ctx, q)
+ if err != nil {
+ return TableCensus{}, err
+ }
+ census := TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now}
+ if retained > 0 {
+ census.TopRoutineID = topID
+ census.TopRoutineShare = float64(topCount) / float64(retained)
+ }
+ return census, nil
+}
+
+// testBetweenRoutineRunCensusReads, when set, runs right after the
+// unknown-status scan and right before routineRunCensusTotals' single-
+// statement read — a deterministic seam for proving a concurrent write
+// landing there cannot desynchronize the retained total from the
+// top-routine attribution, since both now come from that one statement.
+// Never set outside tests.
+var testBetweenRoutineRunCensusReads func(q queryer)
+
+// CensusRuns issues runs' status census, the runs-table equivalent of
+// CensusRoutineRuns without the routine attribution AC-003.5 is specific
+// to office_routine_runs.
+func (s *Store) CensusRuns(ctx context.Context, q queryer, now time.Time) (TableCensus, error) {
+ statusCounts, err := statusCensus(ctx, q, "runs")
+ if err != nil {
+ return TableCensus{}, err
+ }
+ retained, unknown := summarizeStatusCensus(statusCounts, RunHistoryStatuses, RunLiveStatuses)
+ return TableCensus{RetainedCount: retained, UnknownStatuses: unknown, AsOf: now}, nil
+}
+
+// CensusRunEvents is run_events' census: a plain count, since the table
+// has no status column and therefore no unknown-status detection.
+func (s *Store) CensusRunEvents(ctx context.Context, q queryer, now time.Time) (TableCensus, error) {
+ count, err := s.CountRunEvents(ctx, q)
+ if err != nil {
+ return TableCensus{}, err
+ }
+ return TableCensus{RetainedCount: count, AsOf: now}, nil
+}
+
+func statusCensus(ctx context.Context, q queryer, table string) (map[string]int64, error) {
+ query := fmt.Sprintf(`SELECT status, COUNT(*) AS count FROM %s GROUP BY status`, table)
+ rows, err := q.QueryxContext(ctx, query)
+ if err != nil {
+ return nil, err
+ }
+ defer func() { _ = rows.Close() }()
+ counts := map[string]int64{}
+ for rows.Next() {
+ var status string
+ var count int64
+ if err := rows.Scan(&status, &count); err != nil {
+ return nil, err
+ }
+ counts[status] = count
+ }
+ return counts, rows.Err()
+}
+
+// routineRunCensusTotals reads office_routine_runs' table-wide retained
+// total and the routine holding the largest share of it from one
+// statement: a single GROUP BY routine_id pass, with the table total taken
+// as a window sum over that same grouping, so the two numbers reflect
+// exactly one snapshot and a share computed from them can never exceed 1.0.
+// An empty table produces no groups at all; that is the legitimate zero
+// state, not a failure, so sql.ErrNoRows is not propagated.
+func routineRunCensusTotals(ctx context.Context, q queryer) (retained int64, topRoutineID string, topRoutineCount int64, err error) {
+ var row struct {
+ RoutineID string `db:"routine_id"`
+ Retained int64 `db:"retained"`
+ Total int64 `db:"total"`
+ }
+ query := `
+ SELECT routine_id, COUNT(*) AS retained, SUM(COUNT(*)) OVER () AS total
+ FROM office_routine_runs
+ GROUP BY routine_id
+ ORDER BY retained DESC, routine_id ASC
+ LIMIT 1`
+ if err := q.GetContext(ctx, &row, query); err != nil {
+ if errors.Is(err, sql.ErrNoRows) {
+ return 0, "", 0, nil
+ }
+ return 0, "", 0, err
+ }
+ return row.Total, row.RoutineID, row.Retained, nil
+}
diff --git a/apps/backend/internal/office/retention/store_census_test.go b/apps/backend/internal/office/retention/store_census_test.go
new file mode 100644
index 00000000000..a208d79aad8
--- /dev/null
+++ b/apps/backend/internal/office/retention/store_census_test.go
@@ -0,0 +1,235 @@
+package retention
+
+import (
+ "context"
+ "testing"
+ "time"
+
+ "github.com/kandev/kandev/internal/db"
+)
+
+func TestCensusRoutineRuns_RetainedCountIsSumOfEveryStatus(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "skipped", timePtr(daysAgo(2)), daysAgo(2))
+ seedRoutineRun(t, conn, newID(), "r-1", "received", nil, daysAgo(0))
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.RetainedCount != 3 {
+ t.Fatalf("retainedCount = %d, want 3", census.RetainedCount)
+ }
+ if len(census.UnknownStatuses) != 0 {
+ t.Fatalf("unknownStatuses = %v, want none", census.UnknownStatuses)
+ }
+}
+
+func TestCensusRoutineRuns_EmptyTableReturnsZeroNoTopRoutine(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.RetainedCount != 0 {
+ t.Fatalf("retainedCount = %d, want 0", census.RetainedCount)
+ }
+ if census.TopRoutineID != "" {
+ t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID)
+ }
+}
+
+func TestCensusRoutineRuns_DetectsUnknownStatus(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-1", "quarantined", nil, daysAgo(2))
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "quarantined" || census.UnknownStatuses[0].Count != 2 {
+ t.Fatalf("unknownStatuses = %v, want [{quarantined 2}]", census.UnknownStatuses)
+ }
+}
+
+func TestCensusRoutineRuns_AttributesTopRoutineByRetainedShare(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-heavy")
+ seedRoutine(t, conn, "r-light")
+ for i := 0; i < 3; i++ {
+ seedRoutineRun(t, conn, newID(), "r-heavy", "done", timePtr(daysAgo(1)), daysAgo(1))
+ }
+ seedRoutineRun(t, conn, newID(), "r-light", "done", timePtr(daysAgo(1)), daysAgo(1))
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.RetainedCount != 4 {
+ t.Fatalf("retainedCount = %d, want 4", census.RetainedCount)
+ }
+ if census.TopRoutineID != "r-heavy" {
+ t.Fatalf("topRoutineID = %q, want r-heavy", census.TopRoutineID)
+ }
+ if got, want := census.TopRoutineShare, 0.75; got != want {
+ t.Fatalf("topRoutineShare = %v, want %v", got, want)
+ }
+}
+
+func TestCensusRoutineRuns_TiesAttributeToLowerRoutineID(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-b")
+ seedRoutine(t, conn, "r-a")
+ seedRoutineRun(t, conn, newID(), "r-b", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-a", "done", timePtr(daysAgo(1)), daysAgo(1))
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.TopRoutineID != "r-a" {
+ t.Fatalf("topRoutineID = %q, want r-a (lower id on tie)", census.TopRoutineID)
+ }
+}
+
+func TestCensusRuns_RetainedCountIsSumOfEveryStatusNoTopAttribution(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRun(t, conn, newID(), "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1))
+ seedRun(t, conn, newID(), "agent-1", "queued", nil, daysAgo(0))
+ seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0))
+ seedRun(t, conn, newID(), "agent-1", "mystery", nil, daysAgo(0))
+
+ census, err := store.CensusRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRuns: %v", err)
+ }
+ if census.RetainedCount != 4 {
+ t.Fatalf("retainedCount = %d, want 4", census.RetainedCount)
+ }
+ if len(census.UnknownStatuses) != 1 || census.UnknownStatuses[0].Status != "mystery" || census.UnknownStatuses[0].Count != 2 {
+ t.Fatalf("unknownStatuses = %v, want [{mystery 2}]", census.UnknownStatuses)
+ }
+ if census.TopRoutineID != "" {
+ t.Fatalf("topRoutineID = %q, want empty (runs has no routine attribution)", census.TopRoutineID)
+ }
+}
+
+func TestCensusRunEvents_PlainCountNoStatusDetection(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1))
+ seedRunEvent(t, conn, runID, 1)
+ seedRunEvent(t, conn, runID, 2)
+
+ census, err := store.CensusRunEvents(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRunEvents: %v", err)
+ }
+ if census.RetainedCount != 2 {
+ t.Fatalf("retainedCount = %d, want 2", census.RetainedCount)
+ }
+ if census.UnknownStatuses != nil {
+ t.Fatalf("unknownStatuses = %v, want nil", census.UnknownStatuses)
+ }
+}
+
+// TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent
+// proves the fix for the top-routine attribution's former two-query race: a
+// write landing between the unknown-status scan and the single-statement
+// totals read must not let the reported TopRoutineShare and RetainedCount
+// come from different snapshots of the table. Before the fix, this seam sat
+// between two independent reads and could make TopRoutineShare exceed 1.0
+// or attribute a share against a stale total.
+func TestCensusRoutineRuns_ConcurrentWriteBetweenUnknownStatusScanAndTotalsStaysConsistent(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutine(t, conn, "r-2")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+
+ testBetweenRoutineRunCensusReads = func(queryer) {
+ seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1))
+ }
+ t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil })
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+
+ if census.RetainedCount != 3 {
+ t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount)
+ }
+ if census.TopRoutineID != "r-2" {
+ t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID)
+ }
+ if got, want := census.TopRoutineShare, 2.0/3.0; got != want {
+ t.Fatalf("topRoutineShare = %v, want %v", got, want)
+ }
+ if census.TopRoutineShare > 1.0 {
+ t.Fatalf("topRoutineShare = %v, must never exceed 1.0", census.TopRoutineShare)
+ }
+}
+
+// TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError
+// proves routineRunCensusTotals treats a table that became empty as the
+// legitimate zero state rather than propagating sql.ErrNoRows: the old
+// two-query design decided whether to run the top-routine query from a
+// separately-read, now-stale nonzero total, so this same interleaving used
+// to surface an unhandled error instead of a clean zero census.
+func TestCensusRoutineRuns_TableEmptiedBetweenReadsReturnsZeroWithoutError(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+
+ testBetweenRoutineRunCensusReads = func(queryer) {
+ if _, err := conn.Exec(`DELETE FROM office_routine_runs`); err != nil {
+ t.Fatalf("delete all rows: %v", err)
+ }
+ }
+ t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil })
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.RetainedCount != 0 {
+ t.Fatalf("retainedCount = %d, want 0", census.RetainedCount)
+ }
+ if census.TopRoutineID != "" {
+ t.Fatalf("topRoutineID = %q, want empty", census.TopRoutineID)
+ }
+}
+
+func timePtr(t time.Time) *time.Time { return &t }
diff --git a/apps/backend/internal/office/retention/store_postgres_test.go b/apps/backend/internal/office/retention/store_postgres_test.go
new file mode 100644
index 00000000000..91e870e5786
--- /dev/null
+++ b/apps/backend/internal/office/retention/store_postgres_test.go
@@ -0,0 +1,184 @@
+package retention
+
+import (
+ "context"
+ "testing"
+ "time"
+
+ "github.com/jmoiron/sqlx"
+
+ "github.com/kandev/kandev/internal/db"
+ "github.com/kandev/kandev/internal/testutil"
+)
+
+// openSharedSchemaPostgresConn opens a second, independent PostgreSQL
+// connection pointed at the same isolated test schema as an existing
+// connection. A real second physical connection is required to hold a row
+// lock that a concurrent statement on the first connection genuinely blocks
+// on — a sequential in-process test hook cannot reproduce that.
+func openSharedSchemaPostgresConn(t *testing.T, dsn, schema string) *sqlx.DB {
+ t.Helper()
+ raw, err := db.OpenPostgres(dsn, 1, 1)
+ if err != nil {
+ t.Fatalf("open second postgres connection: %v", err)
+ }
+ conn := sqlx.NewDb(raw, "pgx")
+ conn.SetMaxOpenConns(1)
+ conn.SetMaxIdleConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ if _, err := conn.Exec("SET search_path TO " + schema); err != nil {
+ t.Fatalf("set search_path on second connection: %v", err)
+ }
+ return conn
+}
+
+// TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow
+// is the regression test for deleteRunsByIDs' missing direct status/age
+// predicate (AC-OFFICE-RUN-HISTORY-RETENTION-002.4): a run resurrected by a
+// concurrent transaction that has not yet committed when DeleteRunBatch
+// selects it, but commits while the DELETE statement is blocked acquiring
+// the row's lock — driving PostgreSQL's real EvalPlanQual recheck path. The
+// package's existing resurrection tests
+// (TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives) use a
+// sequential in-process hook that resurrects strictly before the DELETE
+// statement starts; they cannot reach this mid-statement window, which
+// needs a second, genuinely concurrent connection.
+func TestDeleteRunBatch_Postgres_ConcurrentResurrectionDuringDeleteExcludesRow(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ sweeper, conn := newPostgresTestSweeper(t, dsn)
+ store := sweeper.store
+
+ var schema string
+ if err := conn.Get(&schema, `SELECT current_schema()`); err != nil {
+ t.Fatalf("select current_schema: %v", err)
+ }
+
+ runID := newID()
+ finished := daysAgo(60)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, runID, 0)
+
+ holder := openSharedSchemaPostgresConn(t, dsn, schema)
+ holderTx, err := holder.BeginTx(ctx, nil)
+ if err != nil {
+ t.Fatalf("begin holder tx: %v", err)
+ }
+ // Resurrecting the row inside an uncommitted transaction takes its row
+ // lock immediately, but the new values are not visible to conn's own
+ // snapshot until Commit below: selectEligibleRunIDs' plain read is
+ // never blocked by an uncommitted writer, so it still sees the row as
+ // terminal and eligible, exactly the window this test targets.
+ if _, err := holderTx.ExecContext(ctx, `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = $1`, runID); err != nil {
+ t.Fatalf("resurrect run inside holder tx: %v", err)
+ }
+
+ admin := testutil.OpenIsolatedPostgres(t, dsn) // separate schema; pg_stat_activity is instance-wide, not schema-scoped
+ deleteDone := make(chan RunBatchResult, 1)
+ deleteErr := make(chan error, 1)
+ go func() {
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ deleteErr <- err
+ return
+ }
+ deleteDone <- result
+ }()
+
+ // Wait for DeleteRunBatch's DELETE statement to actually be blocked on
+ // the holder's row lock before committing, so the recheck this test
+ // targets is guaranteed to happen rather than racing ahead of it.
+ deadline := time.Now().Add(10 * time.Second)
+ for {
+ var waiting bool
+ if err := admin.GetContext(ctx, &waiting, `
+ SELECT EXISTS(
+ SELECT 1 FROM pg_stat_activity
+ WHERE wait_event_type = 'Lock' AND query ILIKE '%DELETE FROM runs%'
+ )`); err != nil {
+ t.Fatalf("poll pg_stat_activity: %v", err)
+ }
+ if waiting {
+ break
+ }
+ if time.Now().After(deadline) {
+ t.Fatal("DeleteRunBatch's DELETE never showed up waiting on the holder's row lock")
+ }
+ time.Sleep(20 * time.Millisecond)
+ }
+
+ if err := holderTx.Commit(); err != nil {
+ t.Fatalf("commit holder tx: %v", err)
+ }
+
+ var result RunBatchResult
+ select {
+ case result = <-deleteDone:
+ case err := <-deleteErr:
+ t.Fatalf("DeleteRunBatch: %v", err)
+ case <-time.After(10 * time.Second):
+ t.Fatal("DeleteRunBatch did not complete after the holder committed")
+ }
+
+ if result.RunsDeleted != 0 || result.Abandoned {
+ t.Fatalf("result = %+v, want a clean no-op: the row was resurrected before the delete's row lock was granted, so PostgreSQL's own recheck of the row must see it as no longer eligible", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run was deleted despite the concurrent recheck")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run's event was deleted despite the concurrent recheck")
+ }
+}
+
+// TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite is
+// AC-OFFICE-RUN-HISTORY-RETENTION-005.1's mandated backlog-path parity test:
+// batch selection order is fixed by named columns
+// (completion_time ASC, id ASC) rather than left to the engine, so this
+// must select and delete the exact same rows on PostgreSQL as
+// TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns proves on
+// SQLite for the identical settings and starting rows.
+func TestDeleteRoutineRunsBatch_Postgres_OldestFirstWithBacklogMatchesSQLite(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ sweeper, conn := newPostgresTestSweeper(t, dsn)
+ store := sweeper.store
+
+ routineID := newID()
+ seedRoutine(t, conn, routineID)
+
+ cutoff := daysAgo(30)
+ oldest, middle, newest := newID(), newID(), newID()
+ oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45)
+ seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC)
+ seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC)
+ seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC)
+
+ // floor 0 so all three are eligible; batch limit 2 -> the two oldest
+ // go, the newest survives as backlog — same as the SQLite test.
+ deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2)
+ if err != nil {
+ t.Fatalf("DeleteRoutineRunsBatch: %v", err)
+ }
+ if deleted != 2 {
+ t.Fatalf("deleted = %d, want 2", deleted)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest); n != 1 {
+ t.Fatal("newest row was deleted on PostgreSQL; oldest-first ordering violated")
+ }
+ for _, id := range []string{oldest, middle} {
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 {
+ t.Fatalf("row %s (older) still present on PostgreSQL after batch limit 2", id)
+ }
+ }
+
+ eligible, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 0)
+ if err != nil {
+ t.Fatalf("CountEligibleRoutineRuns: %v", err)
+ }
+ if eligible != 1 {
+ t.Fatalf("remaining eligible = %d, want 1 (backlog: the newest row is still eligible, just not yet batched)", eligible)
+ }
+}
diff --git a/apps/backend/internal/office/retention/store_test.go b/apps/backend/internal/office/retention/store_test.go
new file mode 100644
index 00000000000..3db5aed1918
--- /dev/null
+++ b/apps/backend/internal/office/retention/store_test.go
@@ -0,0 +1,671 @@
+package retention
+
+import (
+ "context"
+ "testing"
+ "time"
+
+ "github.com/google/uuid"
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/db"
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+)
+
+// testDB builds a fresh in-memory SQLite database carrying the real office
+// schema (including the retention indexes), so eligibility queries run
+// against the genuine table shapes rather than a hand-rolled fixture.
+func testDB(t *testing.T) *sqlx.DB {
+ t.Helper()
+ conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init office schema: %v", err)
+ }
+ return conn
+}
+
+func seedRoutine(t *testing.T, conn *sqlx.DB, id string) {
+ t.Helper()
+ now := time.Now().UTC()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_routines (id, workspace_id, name, created_at, updated_at)
+ VALUES (?, 'ws-1', ?, ?, ?)
+ `), id, id, now, now); err != nil {
+ t.Fatalf("seed routine %s: %v", id, err)
+ }
+}
+
+func seedRoutineRun(t *testing.T, conn *sqlx.DB, id, routineID, status string, completedAt *time.Time, createdAt time.Time) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at)
+ VALUES (?, ?, 'trigger', ?, ?, ?)
+ `), id, routineID, status, completedAt, createdAt); err != nil {
+ t.Fatalf("seed routine run %s: %v", id, err)
+ }
+}
+
+func seedRoutineRunWithFingerprintAndLinkedTask(
+ t *testing.T, conn *sqlx.DB, id, routineID, status string,
+ completedAt *time.Time, createdAt time.Time, fingerprint, linkedTaskID string,
+) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_routine_runs (id, routine_id, source, status, completed_at, created_at, dispatch_fingerprint, linked_task_id)
+ VALUES (?, ?, 'trigger', ?, ?, ?, ?, ?)
+ `), id, routineID, status, completedAt, createdAt, fingerprint, linkedTaskID); err != nil {
+ t.Fatalf("seed routine run %s: %v", id, err)
+ }
+}
+
+func seedRun(t *testing.T, conn *sqlx.DB, id, agentProfileID, status string, finishedAt *time.Time, requestedAt time.Time) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO runs (id, agent_profile_id, reason, status, requested_at, finished_at)
+ VALUES (?, ?, 'test', ?, ?, ?)
+ `), id, agentProfileID, status, requestedAt, finishedAt); err != nil {
+ t.Fatalf("seed run %s: %v", id, err)
+ }
+}
+
+func seedPauseRecovery(t *testing.T, conn *sqlx.DB, agentID, taskID, failedRunID string) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_agent_pause_recoveries (agent_id, task_id, failed_run_id)
+ VALUES (?, ?, ?)
+ `), agentID, taskID, failedRunID); err != nil {
+ t.Fatalf("seed pause recovery for run %s: %v", failedRunID, err)
+ }
+}
+
+func seedRunEvent(t *testing.T, conn *sqlx.DB, runID string, seq int) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO run_events (run_id, seq, event_type, created_at)
+ VALUES (?, ?, 'test', ?)
+ `), runID, seq, time.Now().UTC()); err != nil {
+ t.Fatalf("seed run event for %s: %v", runID, err)
+ }
+}
+
+func seedRouteAttempt(t *testing.T, conn *sqlx.DB, runID string, seq int) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_run_route_attempts (run_id, seq, provider_id, model, tier, outcome, started_at)
+ VALUES (?, ?, 'p', 'm', 't', 'ok', ?)
+ `), runID, seq, time.Now().UTC()); err != nil {
+ t.Fatalf("seed route attempt for %s: %v", runID, err)
+ }
+}
+
+func seedRunSkill(t *testing.T, conn *sqlx.DB, runID, skillID string) {
+ t.Helper()
+ if _, err := conn.Exec(conn.Rebind(`
+ INSERT INTO office_run_skills (run_id, skill_id, version, content_hash, materialized_path)
+ VALUES (?, ?, 'v1', 'hash', '/path')
+ `), runID, skillID); err != nil {
+ t.Fatalf("seed run skill for %s: %v", runID, err)
+ }
+}
+
+func countRows(t *testing.T, conn *sqlx.DB, query string, args ...any) int64 {
+ t.Helper()
+ var n int64
+ if err := conn.Get(&n, conn.Rebind(query), args...); err != nil {
+ t.Fatalf("count query %q: %v", query, err)
+ }
+ return n
+}
+
+func daysAgo(n int) time.Time { return time.Now().UTC().AddDate(0, 0, -n) }
+
+func newID() string { return uuid.New().String() }
+
+func TestCountEligibleRoutineRuns_RespectsStatusWindowAndFloor(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+ routineID := newID()
+ seedRoutine(t, conn, routineID)
+
+ cutoff := daysAgo(30)
+ // 3 history rows older than the window, floor 50: all inside the floor,
+ // none eligible.
+ for i := 0; i < 3; i++ {
+ completed := daysAgo(40 + i)
+ seedRoutineRun(t, conn, newID(), routineID, "coalesced", &completed, completed)
+ }
+ // A task_created (live-state) row, ancient: never eligible regardless of
+ // age (AC-OFFICE-RUN-HISTORY-RETENTION-001.1).
+ ancient := daysAgo(3650)
+ seedRoutineRun(t, conn, newID(), routineID, "task_created", nil, ancient)
+
+ count, err := store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50)
+ if err != nil {
+ t.Fatalf("CountEligibleRoutineRuns: %v", err)
+ }
+ if count != 0 {
+ t.Fatalf("count = %d, want 0 (all 3 history rows within the floor of 50)", count)
+ }
+
+ // Add 50 more, all older than window: total history rows now 53, floor
+ // 50, so exactly 3 are eligible (the 3 oldest, since floor keeps the
+ // newest 50 of the 53).
+ for i := 0; i < 50; i++ {
+ completed := daysAgo(35 + i)
+ seedRoutineRun(t, conn, newID(), routineID, "done", &completed, completed)
+ }
+ count, err = store.CountEligibleRoutineRuns(ctx, conn, cutoff, 50)
+ if err != nil {
+ t.Fatalf("CountEligibleRoutineRuns: %v", err)
+ }
+ if count != 3 {
+ t.Fatalf("count = %d, want 3 (53 history rows, floor 50)", count)
+ }
+}
+
+func TestDeleteRoutineRunsBatch_OldestFirstAndOrderedByNamedColumns(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+ routineID := newID()
+ seedRoutine(t, conn, routineID)
+
+ cutoff := daysAgo(30)
+ var oldest, middle, newest string
+ oldest, middle, newest = newID(), newID(), newID()
+ oldC, midC, newC := daysAgo(90), daysAgo(60), daysAgo(45)
+ seedRoutineRun(t, conn, oldest, routineID, "done", &oldC, oldC)
+ seedRoutineRun(t, conn, middle, routineID, "done", &midC, midC)
+ seedRoutineRun(t, conn, newest, routineID, "done", &newC, newC)
+
+ // floor 0 so all three are eligible; batch limit 2 -> the two oldest go,
+ // the newest survives as backlog.
+ deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 0, 2)
+ if err != nil {
+ t.Fatalf("DeleteRoutineRunsBatch: %v", err)
+ }
+ if deleted != 2 {
+ t.Fatalf("deleted = %d, want 2", deleted)
+ }
+ remaining := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newest)
+ if remaining != 1 {
+ t.Fatalf("newest row was deleted; oldest-first ordering violated")
+ }
+ for _, id := range []string{oldest, middle} {
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, id); n != 0 {
+ t.Fatalf("row %s (older) still present after batch limit 2", id)
+ }
+ }
+}
+
+func TestDeleteRoutineRunsBatch_FloorReassertedAtDeleteTime(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+ routineID := newID()
+ seedRoutine(t, conn, routineID)
+
+ cutoff := daysAgo(30)
+ c := daysAgo(90)
+ id := newID()
+ seedRoutineRun(t, conn, id, routineID, "done", &c, c)
+
+ // Floor 1 (>= the single row present) means the row is protected: it
+ // is the newest (and only) row for its routine.
+ deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 1, 100)
+ if err != nil {
+ t.Fatalf("DeleteRoutineRunsBatch: %v", err)
+ }
+ if deleted != 0 {
+ t.Fatalf("deleted = %d, want 0 (row is inside the floor)", deleted)
+ }
+}
+
+// TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-001.4's floor is per-owner: every seed
+// helper elsewhere in this suite uses a single routine, so a regression
+// that dropped routineRunEligibleSubquery's PARTITION BY routine_id
+// (turning a per-owner floor into one shared across every routine) would
+// otherwise leave the whole suite green.
+func TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ routineA, routineB := newID(), newID()
+ seedRoutine(t, conn, routineA)
+ seedRoutine(t, conn, routineB)
+
+ cutoff := daysAgo(30)
+ var newestA, newestB string
+ for i := 0; i < 5; i++ {
+ completed := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86)
+ idA, idB := newID(), newID()
+ seedRoutineRun(t, conn, idA, routineA, "done", &completed, completed)
+ seedRoutineRun(t, conn, idB, routineB, "done", &completed, completed)
+ if i == 4 {
+ newestA, newestB = idA, idB
+ }
+ }
+
+ // Floor 3 per routine: each routine has 5 history rows, so 2 are
+ // eligible per routine, 4 total. A floor shared across both routines
+ // (10 rows, floor 3) would instead delete 7 and could delete either
+ // routine's newest row.
+ deleted, err := store.DeleteRoutineRunsBatch(ctx, conn, cutoff, 3, 100)
+ if err != nil {
+ t.Fatalf("DeleteRoutineRunsBatch: %v", err)
+ }
+ if deleted != 4 {
+ t.Fatalf("deleted = %d, want 4 (2 eligible per routine, floor 3 held independently)", deleted)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineA); n != 3 {
+ t.Fatalf("routineA remaining = %d, want 3 (its own floor)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE routine_id = ?`, routineB); n != 3 {
+ t.Fatalf("routineB remaining = %d, want 3 (its own floor)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestA); n != 1 {
+ t.Fatalf("routineA's newest row was deleted; its floor should have protected it")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, newestB); n != 1 {
+ t.Fatalf("routineB's newest row was deleted; its floor should have protected it")
+ }
+}
+
+func TestDeleteRunBatch_DeletesSatellitesAtomicallyWithRun(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ finished := daysAgo(60)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, runID, 0)
+ seedRunEvent(t, conn, runID, 1)
+ seedRouteAttempt(t, conn, runID, 0)
+ seedRunSkill(t, conn, runID, "skill-1")
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 1 || result.RunEventsDeleted != 2 || result.RouteAttemptsDeleted != 1 || result.RunSkillsDeleted != 1 {
+ t.Fatalf("result = %+v, want RunsDeleted=1 RunEventsDeleted=2 RouteAttemptsDeleted=1 RunSkillsDeleted=1", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 0 {
+ t.Fatalf("%d run_events rows remain referencing a deleted run", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_route_attempts WHERE run_id = ?`, runID); n != 0 {
+ t.Fatalf("%d route attempt rows remain referencing a deleted run", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_run_skills WHERE run_id = ?`, runID); n != 0 {
+ t.Fatalf("%d run skill rows remain referencing a deleted run", n)
+ }
+}
+
+// TestRunRetention_PreservesActivePauseRecoveryRun proves that a failed run
+// still referenced by MarkAgentPausedFixed remains available until its
+// recovery snapshot is consumed or discarded. Count and delete must use the
+// same protection predicate so a preview cannot promise deletion that the
+// batch path applies.
+func TestRunRetention_PreservesActivePauseRecoveryRun(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ failedAt := daysAgo(60)
+ seedRun(t, conn, runID, "agent-recovery", "failed", &failedAt, failedAt)
+ seedPauseRecovery(t, conn, "agent-recovery", "task-recovery", runID)
+
+ cutoff := daysAgo(30)
+ count, err := store.CountEligibleRuns(ctx, conn, cutoff, 0)
+ if err != nil {
+ t.Fatalf("CountEligibleRuns: %v", err)
+ }
+ if count != 0 {
+ t.Fatalf("eligible count = %d, want 0 while pause recovery references the run", count)
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, cutoff, 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 0 || result.Abandoned {
+ t.Fatalf("result = %+v, want a clean no-op while recovery is active", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 {
+ t.Fatalf("recovery run count = %d, want 1", n)
+ }
+
+ if _, err := conn.Exec(conn.Rebind(
+ `DELETE FROM office_agent_pause_recoveries WHERE agent_id = ? AND task_id = ?`,
+ ), "agent-recovery", "task-recovery"); err != nil {
+ t.Fatalf("discard pause recovery: %v", err)
+ }
+
+ count, err = store.CountEligibleRuns(ctx, conn, cutoff, 0)
+ if err != nil {
+ t.Fatalf("CountEligibleRuns after recovery discard: %v", err)
+ }
+ if count != 1 {
+ t.Fatalf("eligible count after recovery discard = %d, want 1", count)
+ }
+ result, err = store.DeleteRunBatch(ctx, conn, cutoff, 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch after recovery discard: %v", err)
+ }
+ if result.RunsDeleted != 1 || result.Abandoned {
+ t.Fatalf("result after recovery discard = %+v, want one deleted run", result)
+ }
+}
+
+// TestDeleteRunBatch_SurvivingRunKeepsEveryEvent proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-001.6: run_events is never deleted for a
+// run that is not itself being deleted in the same transaction, no matter
+// how old those events are.
+func TestDeleteRunBatch_SurvivingRunKeepsEveryEvent(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ liveRunID := newID()
+ // queued: live state, never eligible at any age.
+ seedRun(t, conn, liveRunID, "agent-1", "queued", nil, daysAgo(400))
+ for i := 0; i < 5; i++ {
+ seedRunEvent(t, conn, liveRunID, i)
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 0 {
+ t.Fatalf("RunsDeleted = %d, want 0 (queued run is live state)", result.RunsDeleted)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, liveRunID); n != 5 {
+ t.Fatalf("run_events for a surviving run = %d, want 5 untouched", n)
+ }
+}
+
+// TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected proves a
+// run resurrected to "queued" (as ScheduleRetry does, clearing finished_at)
+// before DeleteRunBatch runs at all is excluded by the eligibility
+// selection itself, leaving it and its satellites untouched. This is the
+// simple case; the delete-time re-assertion this package's AC-002.4
+// re-assertion actually catches — a resurrection landing between
+// selection and the delete statement, inside one attempt — is covered by
+// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives below.
+func TestDeleteRunBatch_RunResurrectedBeforeSelectionIsNeverSelected(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ finished := daysAgo(60)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, runID, 0)
+
+ // Simulate the resurrection race by racing a resurrecting UPDATE
+ // against DeleteRunBatch using a second connection to the same
+ // in-memory database (SQLite's single-writer serializes them, but the
+ // delete's re-assertion inside the transaction is what must catch the
+ // now-live row regardless of interleaving — resurrecting up front is
+ // the deterministic way to exercise that same code path without
+ // depending on goroutine scheduling).
+ if _, err := conn.Exec(conn.Rebind(`
+ UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?
+ `), runID); err != nil {
+ t.Fatalf("resurrect run: %v", err)
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 0 || result.Abandoned {
+ t.Fatalf("result = %+v, want a clean no-op (resurrected run was never selected)", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run was deleted")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run's event was deleted")
+ }
+}
+
+// TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives drives
+// the retry path deterministically via testAfterSelectEligibleRunIDs: the
+// row is eligible at selection time (so it enters attempt 0's id list),
+// resurrected by the hook immediately after that selection (before
+// deleteRunBatchOnce's own DELETE runs), so the re-assertion inside the
+// transaction detects the mismatch, rolls back, and attempt 1 re-selects
+// against the now-live row and finds nothing to do. The run and its
+// satellite survive throughout.
+func TestDeleteRunBatch_MidTransactionResurrectionRetriesThenSurvives(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ finished := daysAgo(60)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, runID, 0)
+
+ t.Cleanup(func() { testAfterSelectEligibleRunIDs = nil })
+ testAfterSelectEligibleRunIDs = func(attempt int, ids []string) {
+ if attempt != 0 {
+ return
+ }
+ if _, err := conn.Exec(conn.Rebind(
+ `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`,
+ ), runID); err != nil {
+ t.Fatalf("resurrect run mid-transaction: %v", err)
+ }
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 0 || result.Abandoned {
+ t.Fatalf("result = %+v, want a clean no-op after the retry re-selects nothing", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run was deleted despite the retry")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 {
+ t.Fatalf("resurrected run's event was deleted despite the retry")
+ }
+}
+
+// TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches forces the
+// resurrection race to land on *both* attempts via
+// testAfterSelectEligibleRunIDs: the row is repeatedly resurrected right
+// after each selection, so both attempts' delete re-assertions mismatch.
+// DeleteRunBatch must give up rather than loop forever, reporting the
+// batch as Abandoned — which the sweep records as that table's failure,
+// not backlog (AC-OFFICE-RUN-HISTORY-RETENTION-002.7).
+func TestDeleteRunBatch_AbandonsAfterTwoConsecutiveMismatches(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ runID := newID()
+ finished := daysAgo(60)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, runID, 0)
+
+ t.Cleanup(func() {
+ testBeforeSelectEligibleRunIDs = nil
+ testAfterSelectEligibleRunIDs = nil
+ })
+ beforeCalls, afterCalls := 0, 0
+ // Before each attempt's selection, make sure the row reads terminal
+ // again (undoing the previous attempt's resurrection) so it is
+ // selected every time, not just on attempt 0.
+ testBeforeSelectEligibleRunIDs = func(attempt int) {
+ beforeCalls++
+ if attempt == 0 {
+ return // already terminal from seeding
+ }
+ if _, err := conn.Exec(conn.Rebind(
+ `UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`,
+ ), finished, runID); err != nil {
+ t.Fatalf("re-terminalize run before attempt %d: %v", attempt, err)
+ }
+ }
+ // After each attempt's selection (which just proved the row was
+ // terminal), flip it live so that attempt's own delete re-assertion
+ // mismatches.
+ testAfterSelectEligibleRunIDs = func(attempt int, ids []string) {
+ afterCalls++
+ if _, err := conn.Exec(conn.Rebind(
+ `UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`,
+ ), runID); err != nil {
+ t.Fatalf("resurrect run mid-transaction (attempt %d): %v", attempt, err)
+ }
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if !result.Abandoned {
+ t.Fatalf("result = %+v, want Abandoned=true after two consecutive mismatches", result)
+ }
+ if result.RunsDeleted != 0 || result.RunEventsDeleted != 0 {
+ t.Fatalf("result = %+v, want every count zero on an abandoned batch", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, runID); n != 1 {
+ t.Fatalf("run was deleted despite an abandoned batch")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events WHERE run_id = ?`, runID); n != 1 {
+ t.Fatalf("run_events was deleted despite an abandoned batch")
+ }
+ if beforeCalls != 2 || afterCalls != 2 {
+ t.Fatalf("before/after hooks invoked %d/%d times, want exactly 2/2 (one per attempt)", beforeCalls, afterCalls)
+ }
+}
+
+// TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile is the runs-table
+// equivalent of TestDeleteRoutineRunsBatch_FloorHeldIndependentlyPerRoutine:
+// every other DeleteRunBatch test in this file uses a single
+// "agent-1" owner, so a regression that dropped runEligibleSubquery's
+// PARTITION BY agent_profile_id would otherwise leave the whole suite green.
+func TestDeleteRunBatch_FloorHeldIndependentlyPerAgentProfile(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ cutoff := daysAgo(30)
+ var newestA, newestB string
+ for i := 0; i < 5; i++ {
+ finished := daysAgo(90 - i) // i=0 oldest (day 90) .. i=4 newest (day 86)
+ idA, idB := newID(), newID()
+ seedRun(t, conn, idA, "agent-a", "finished", &finished, finished)
+ seedRun(t, conn, idB, "agent-b", "finished", &finished, finished)
+ if i == 4 {
+ newestA, newestB = idA, idB
+ }
+ }
+
+ // Floor 3 per agent profile: each owns 5 history rows, so 2 are
+ // eligible per owner, 4 total.
+ result, err := store.DeleteRunBatch(ctx, conn, cutoff, 3, 100)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.RunsDeleted != 4 || result.Abandoned {
+ t.Fatalf("result = %+v, want 4 deleted (2 eligible per agent profile, floor 3 held independently)", result)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-a"); n != 3 {
+ t.Fatalf("agent-a remaining = %d, want 3 (its own floor)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE agent_profile_id = ?`, "agent-b"); n != 3 {
+ t.Fatalf("agent-b remaining = %d, want 3 (its own floor)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestA); n != 1 {
+ t.Fatalf("agent-a's newest run was deleted; its floor should have protected it")
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, newestB); n != 1 {
+ t.Fatalf("agent-b's newest run was deleted; its floor should have protected it")
+ }
+}
+
+// TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements proves
+// deleteRunBatchOnce's satellite and runs deletes split their id list into
+// chunks of at most retentionMaxHostParams rather than binding the whole
+// batch as one IN clause, which is what let batch_limit's documented range
+// (AC-OFFICE-RUN-HISTORY-RETENTION-004.3, up to 100,000) overflow a single
+// statement's bind-parameter limit on either engine. retentionMaxHostParams+2
+// runs, each with one satellite row apiece, forces the delete loop to span
+// more than one chunk; every row and every satellite must still be deleted
+// and the reported count must reflect the true total, not just one chunk's.
+func TestDeleteRunBatch_LargeBatchChunksIDListAcrossStatements(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+
+ const rowCount = retentionMaxHostParams + 2
+ finished := daysAgo(60)
+ ids := make([]string, 0, rowCount)
+ for i := 0; i < rowCount; i++ {
+ id := newID()
+ seedRun(t, conn, id, "agent-1", "finished", &finished, finished)
+ seedRunEvent(t, conn, id, 0)
+ ids = append(ids, id)
+ }
+
+ result, err := store.DeleteRunBatch(ctx, conn, daysAgo(30), 0, rowCount)
+ if err != nil {
+ t.Fatalf("DeleteRunBatch: %v", err)
+ }
+ if result.Abandoned {
+ t.Fatalf("result = %+v, want a clean delete, not abandoned", result)
+ }
+ if result.RunsDeleted != int64(rowCount) {
+ t.Fatalf("RunsDeleted = %d, want %d", result.RunsDeleted, rowCount)
+ }
+ if result.RunEventsDeleted != int64(rowCount) {
+ t.Fatalf("RunEventsDeleted = %d, want %d", result.RunEventsDeleted, rowCount)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 {
+ t.Fatalf("runs remaining = %d, want 0 (every row across every chunk must be deleted)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM run_events`); n != 0 {
+ t.Fatalf("run_events remaining = %d, want 0", n)
+ }
+ for _, id := range ids {
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, id); n != 0 {
+ t.Fatalf("run %s remains after a chunked delete", id)
+ }
+ }
+}
+
+func TestCountRunEvents_PlainCount(t *testing.T) {
+ conn := testDB(t)
+ store := NewStore(db.NewPool(conn, conn))
+ ctx := context.Background()
+ runID := newID()
+ finished := daysAgo(1)
+ seedRun(t, conn, runID, "agent-1", "finished", &finished, finished)
+ for i := 0; i < 4; i++ {
+ seedRunEvent(t, conn, runID, i)
+ }
+ count, err := store.CountRunEvents(ctx, conn)
+ if err != nil {
+ t.Fatalf("CountRunEvents: %v", err)
+ }
+ if count != 4 {
+ t.Fatalf("CountRunEvents() = %d, want 4", count)
+ }
+}
diff --git a/apps/backend/internal/office/retention/sweep.go b/apps/backend/internal/office/retention/sweep.go
new file mode 100644
index 00000000000..9c972d19cf4
--- /dev/null
+++ b/apps/backend/internal/office/retention/sweep.go
@@ -0,0 +1,331 @@
+package retention
+
+import (
+ "context"
+ "sync"
+ "time"
+
+ "github.com/kandev/kandev/internal/db"
+)
+
+// TableSweepResult is one reported table's outcome for one sweep
+// (AC-OFFICE-RUN-HISTORY-RETENTION-004.6). Backlog is only ever true for a
+// swept table's own SweptTableResult; a satellite table's Backlog stays
+// false because its deletion is never independently batch-limited — it
+// deletes exactly the run ids its owning runs batch selected.
+type TableSweepResult struct {
+ Deleted int64 `json:"deleted"`
+ Backlog bool `json:"backlog"`
+ Err string `json:"error"`
+}
+
+// SweptTableResult adds preview reporting to a swept table's outcome.
+// Previewed is true only when THIS sweep was a preview pass for this
+// table, never a running total.
+type SweptTableResult struct {
+ TableSweepResult
+ Previewed bool `json:"previewed"`
+ WouldDelete int64 `json:"would_delete"`
+}
+
+// LastSweep is the in-memory value replaced wholesale at the end of each
+// sweep that actually ran (AC-OFFICE-RUN-HISTORY-RETENTION-004.6, -004.7).
+// Nothing here is persisted.
+type LastSweep struct {
+ StartedAt time.Time `json:"started_at"`
+ FinishedAt time.Time `json:"finished_at"`
+
+ OfficeRoutineRuns SweptTableResult `json:"office_routine_runs"`
+ Runs SweptTableResult `json:"runs"`
+ RunEvents TableSweepResult `json:"run_events"`
+ RouteAttempts TableSweepResult `json:"route_attempts"`
+ RunSkills TableSweepResult `json:"run_skills"`
+}
+
+type satelliteResults struct {
+ RunEvents TableSweepResult
+ RouteAttempts TableSweepResult
+ RunSkills TableSweepResult
+}
+
+// Sweeper runs one sweep or one census pass at a time; the scheduler
+// (scheduler.go) owns when to call each.
+//
+// - Lost-exclusivity outcome: a lock lost between tables abandons the
+// whole sweep attempt as a skip — the same outcome as a local
+// concurrent-sweep collision — rather than inventing a per-table
+// "skipped" state the design's eight-id health catalogue has nowhere
+// to report. Batches already committed on PostgreSQL stay committed
+// (AC-002.5); LastSweep simply is not replaced by this attempt, and
+// the next scheduled sweep reports the tables' true state either way.
+// - The skip counter increments for a collision, a failed settings
+// re-read, a failed lock acquisition, and a lock lost mid-sweep — every
+// case where a sweep was due and did not produce a result. It does
+// NOT increment when retention is disabled (AC-002.8's "run no sweep"
+// is a deliberate, steady-state condition, not a due sweep that could
+// not run; incrementing forever while off would make the counter
+// meaningless).
+type Sweeper struct {
+ pool *db.Pool
+ store *Store
+ settingsStore *SettingsStore
+ previewMarker *PreviewMarkerStore
+ census *CensusTracker
+
+ mu sync.Mutex
+ sweeping bool
+ lastSweep *LastSweep
+ skipCount int64
+ lastSkip time.Time
+}
+
+// NewSweeper wires the sweep orchestration to its dependencies.
+func NewSweeper(pool *db.Pool, store *Store, settingsStore *SettingsStore, previewMarker *PreviewMarkerStore) *Sweeper {
+ return &Sweeper{
+ pool: pool,
+ store: store,
+ settingsStore: settingsStore,
+ previewMarker: previewMarker,
+ census: NewCensusTracker(),
+ }
+}
+
+// LastSweepSnapshot returns the most recent completed sweep's result. ok is
+// false before the first sweep ever completes (AC-OFFICE-RUN-HISTORY-RETENTION-004.7).
+func (s *Sweeper) LastSweepSnapshot() (LastSweep, bool) {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ if s.lastSweep == nil {
+ return LastSweep{}, false
+ }
+ return *s.lastSweep, true
+}
+
+// SkipSnapshot returns the running skip count and the last skip's time.
+func (s *Sweeper) SkipSnapshot() (count int64, lastAt time.Time) {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ return s.skipCount, s.lastSkip
+}
+
+// CensusSnapshot returns the current per-table retained counts.
+func (s *Sweeper) CensusSnapshot() RetainedCounts {
+ return s.census.Snapshot()
+}
+
+// RunSweep performs at most one sweep: office_routine_runs, then runs with
+// its satellites, in that fixed order (AC-OFFICE-RUN-HISTORY-RETENTION-002.6).
+func (s *Sweeper) RunSweep(ctx context.Context) {
+ if !s.beginSweep() {
+ s.recordSkip()
+ return
+ }
+ defer s.endSweep()
+
+ settings, err := s.settingsStore.GetSettingsForSweep(ctx)
+ if err != nil {
+ // Settings unreadable at sweep start fails closed: no sweep, no
+ // fallback to the (possibly shorter) default window.
+ s.recordSkip()
+ return
+ }
+ if !settings.Enabled {
+ // AC-002.8: disabled means no sweep at all, not a recorded skip.
+ return
+ }
+
+ q, session, ok := s.acquireQueryer(ctx)
+ if !ok {
+ s.recordSkip()
+ return
+ }
+ if session != nil {
+ defer session.release()
+ }
+
+ now := time.Now().UTC()
+ report := LastSweep{StartedAt: now}
+ report.OfficeRoutineRuns = s.sweepRoutineRuns(ctx, q, settings.RoutineRuns, now, settings.BatchLimit)
+
+ if testBetweenTablesSweep != nil {
+ testBetweenTablesSweep(q)
+ }
+ if session != nil && !session.alive(ctx) {
+ // Exclusivity was lost after the first table's work. Do not
+ // start the second table, and do not publish a partial result —
+ // this whole attempt is a skip, exactly as if the lock had
+ // never been acquired.
+ s.recordSkip()
+ return
+ }
+
+ runsResult, satellites := s.sweepRuns(ctx, q, settings.Runs, now, settings.BatchLimit)
+ report.Runs = runsResult
+ report.RunEvents = satellites.RunEvents
+ report.RouteAttempts = satellites.RouteAttempts
+ report.RunSkills = satellites.RunSkills
+
+ report.FinishedAt = time.Now().UTC()
+ s.mu.Lock()
+ s.lastSweep = &report
+ s.mu.Unlock()
+
+ incSweepCompleted()
+ incDeleted(TableOfficeRoutineRuns, report.OfficeRoutineRuns.Deleted)
+ incDeleted(TableRuns, report.Runs.Deleted)
+ incDeleted(TableRunEvents, report.RunEvents.Deleted)
+ incDeleted("office_run_route_attempts", report.RouteAttempts.Deleted)
+ incDeleted("office_run_skills", report.RunSkills.Deleted)
+}
+
+// testBetweenTablesSweep, when set, runs right after office_routine_runs'
+// table work and right before the alive() liveness check and the runs
+// table — a deterministic seam for exercising AC-OFFICE-RUN-HISTORY-RETENTION-002.12's
+// "verifies the lock connection is still alive between tables" path
+// without depending on real cross-process timing. It receives the sweep's
+// own queryer so a test can run diagnostics (or a second sweep attempt)
+// against the exact connection in use. Never set outside tests.
+var testBetweenTablesSweep func(q queryer)
+
+// RunCensus evaluates the retained-count census for every thresholded
+// table. Read-only, so it needs no advisory lock: every backend computes
+// and serves its own local view.
+func (s *Sweeper) RunCensus(ctx context.Context) {
+ q := s.pool.Reader()
+ now := time.Now().UTC()
+
+ routineCensus, err := s.store.CensusRoutineRuns(ctx, q, now)
+ s.census.RecordRoutineRuns(routineCensus, err)
+ incCensus(TableOfficeRoutineRuns, err)
+
+ runsCensus, err := s.store.CensusRuns(ctx, q, now)
+ s.census.RecordRuns(runsCensus, err)
+ incCensus(TableRuns, err)
+
+ runEventsCensus, err := s.store.CensusRunEvents(ctx, q, now)
+ s.census.RecordRunEvents(runEventsCensus, err)
+ incCensus(TableRunEvents, err)
+}
+
+func (s *Sweeper) beginSweep() bool {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ if s.sweeping {
+ return false
+ }
+ s.sweeping = true
+ return true
+}
+
+func (s *Sweeper) endSweep() {
+ s.mu.Lock()
+ s.sweeping = false
+ s.mu.Unlock()
+}
+
+func (s *Sweeper) recordSkip() {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ s.skipCount++
+ s.lastSkip = time.Now().UTC()
+ incSweepSkipped()
+}
+
+func (s *Sweeper) acquireQueryer(ctx context.Context) (queryer, *sweepSession, bool) {
+ if !s.store.IsPostgres() {
+ return s.pool.Writer(), nil, true
+ }
+ session, ok, err := acquireSweepSession(ctx, s.pool)
+ if err != nil || !ok {
+ return nil, nil, false
+ }
+ return session.queryer(), session, true
+}
+
+// isPreviewed reads the preview marker through q, the sweep's own
+// connection, rather than the shared settings pool (see the
+// PreviewMarkerStore.GetWith doc comment).
+func (s *Sweeper) isPreviewed(ctx context.Context, q queryer, table TableName) bool {
+ marker, _ := s.previewMarker.GetWith(ctx, q)
+ _, ok := marker[table]
+ return ok
+}
+
+func (s *Sweeper) sweepRoutineRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) SweptTableResult {
+ cutoff := retentionCutoff(now, cfg.WindowDays)
+ if !s.isPreviewed(ctx, q, TableOfficeRoutineRuns) {
+ return s.previewTable(ctx, q, TableOfficeRoutineRuns, func() (int64, error) {
+ return s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner)
+ }, now)
+ }
+
+ eligible, err := s.store.CountEligibleRoutineRuns(ctx, q, cutoff, cfg.FloorPerOwner)
+ if err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}
+ }
+ deleted, err := s.store.DeleteRoutineRunsBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit)
+ if err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}
+ }
+ return SweptTableResult{TableSweepResult: TableSweepResult{
+ Deleted: deleted,
+ Backlog: eligible > int64(batchLimit),
+ }}
+}
+
+func (s *Sweeper) sweepRuns(ctx context.Context, q queryer, cfg TableSettings, now time.Time, batchLimit int) (SweptTableResult, satelliteResults) {
+ cutoff := retentionCutoff(now, cfg.WindowDays)
+ if !s.isPreviewed(ctx, q, TableRuns) {
+ result := s.previewTable(ctx, q, TableRuns, func() (int64, error) {
+ return s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner)
+ }, now)
+ return result, satelliteResults{}
+ }
+
+ eligible, err := s.store.CountEligibleRuns(ctx, q, cutoff, cfg.FloorPerOwner)
+ if err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{}
+ }
+ result, err := s.store.DeleteRunBatch(ctx, q, cutoff, cfg.FloorPerOwner, batchLimit)
+ if err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}, satelliteResults{}
+ }
+ if result.Abandoned {
+ // AC-002.7: an abandoned batch is that table's failure, not
+ // backlog; the satellites report zero, matching what the
+ // rollback actually left committed.
+ return SweptTableResult{TableSweepResult: TableSweepResult{
+ Err: "batch abandoned after two consecutive mismatches",
+ }}, satelliteResults{}
+ }
+ return SweptTableResult{TableSweepResult: TableSweepResult{
+ Deleted: result.RunsDeleted,
+ Backlog: eligible > int64(batchLimit),
+ }}, satelliteResults{
+ RunEvents: TableSweepResult{Deleted: result.RunEventsDeleted},
+ RouteAttempts: TableSweepResult{Deleted: result.RouteAttemptsDeleted},
+ RunSkills: TableSweepResult{Deleted: result.RunSkillsDeleted},
+ }
+}
+
+// retentionCutoff turns a table's configured window into the instant a
+// history row's completion time must be older than to be eligible,
+// derived from the one sweep-start instant both tables share
+// (AC-OFFICE-RUN-HISTORY-RETENTION-002.11).
+func retentionCutoff(now time.Time, windowDays int) time.Time {
+ return now.AddDate(0, 0, -windowDays)
+}
+
+// previewTable runs a table's first-ever preview pass: count eligible rows
+// uncapped, mark the table previewed, and delete nothing
+// (AC-OFFICE-RUN-HISTORY-RETENTION-003.2, -003.9).
+func (s *Sweeper) previewTable(ctx context.Context, q queryer, table TableName, countEligible func() (int64, error), now time.Time) SweptTableResult {
+ count, err := countEligible()
+ if err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}
+ }
+ if err := s.previewMarker.MarkCompletedWith(ctx, q, table, now); err != nil {
+ return SweptTableResult{TableSweepResult: TableSweepResult{Err: err.Error()}}
+ }
+ return SweptTableResult{Previewed: true, WouldDelete: count}
+}
diff --git a/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go
new file mode 100644
index 00000000000..7c9ed78b9a6
--- /dev/null
+++ b/apps/backend/internal/office/retention/sweep_lookup_integrity_test.go
@@ -0,0 +1,113 @@
+package retention
+
+import (
+ "context"
+ "testing"
+
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+)
+
+// TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-005.3: after a sweep deletes unrelated
+// history rows, including ones sharing the same routine and fingerprint,
+// the concurrency gate's fingerprint lookup still returns exactly the live
+// task_created run it would have returned had no sweep run.
+func TestGetActiveRunForFingerprint_UnaffectedByRetentionSweep(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ repo, err := officesqlite.NewWithDB(conn, conn, nil)
+ if err != nil {
+ t.Fatalf("open repo: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "done", &old, old, "fp-1", "")
+ seedRoutineRunWithFingerprintAndLinkedTask(t, conn, newID(), "r-1", "failed", &old, old, "fp-1", "")
+
+ activeID := newID()
+ seedRoutineRunWithFingerprintAndLinkedTask(t, conn, activeID, "r-1", "task_created", nil, daysAgo(1), "fp-1", "")
+
+ before, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1")
+ if err != nil {
+ t.Fatalf("GetActiveRunForFingerprint (before): %v", err)
+ }
+ if before == nil || before.ID != activeID {
+ t.Fatalf("GetActiveRunForFingerprint (before) = %+v, want id %s", before, activeID)
+ }
+
+ sweeper.RunSweep(ctx) // preview pass
+ sweeper.RunSweep(ctx) // deleting pass
+
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE status IN ('done', 'failed')`); n != 0 {
+ t.Fatalf("old rows remaining = %d, want 0 (sweep should have deleted them)", n)
+ }
+
+ after, err := repo.GetActiveRunForFingerprint(ctx, "r-1", "fp-1")
+ if err != nil {
+ t.Fatalf("GetActiveRunForFingerprint (after): %v", err)
+ }
+ if after == nil || after.ID != activeID {
+ t.Fatalf("GetActiveRunForFingerprint (after sweep) = %+v, want the same active run %s", after, activeID)
+ }
+}
+
+// TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing proves
+// AC-OFFICE-RUN-HISTORY-RETENTION-005.4: once a sweep deletes the routine
+// run linked to a task, the task-closure lookup for that task ID resolves
+// to nothing rather than to an unrelated run.
+func TestGetRoutineRunByLinkedTaskID_DeletedRunResolvesToNothing(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ repo, err := officesqlite.NewWithDB(conn, conn, nil)
+ if err != nil {
+ t.Fatalf("open repo: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ deletedID := newID()
+ seedRoutineRunWithFingerprintAndLinkedTask(t, conn, deletedID, "r-1", "done", &old, old, "fp-1", "task-1")
+
+ // A second, unrelated run under a different task must not be
+ // mistaken for task-1's closed-out run.
+ other := daysAgo(1)
+ otherID := newID()
+ seedRoutineRunWithFingerprintAndLinkedTask(t, conn, otherID, "r-1", "done", &other, other, "fp-2", "task-2")
+
+ before, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1")
+ if err != nil {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (before): %v", err)
+ }
+ if before == nil || before.ID != deletedID {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (before) = %+v, want id %s", before, deletedID)
+ }
+
+ sweeper.RunSweep(ctx) // preview pass
+ sweeper.RunSweep(ctx) // deleting pass
+
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, deletedID); n != 0 {
+ t.Fatalf("deleted run rows = %d, want 0", n)
+ }
+
+ after, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-1")
+ if err != nil {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (after): %v", err)
+ }
+ if after != nil {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (after sweep) = %+v, want nil", after)
+ }
+
+ // task-2's own run must be unaffected by task-1's deletion.
+ stillThere, err := repo.GetRoutineRunByLinkedTaskID(ctx, "task-2")
+ if err != nil {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (task-2): %v", err)
+ }
+ if stillThere == nil || stillThere.ID != otherID {
+ t.Fatalf("GetRoutineRunByLinkedTaskID (task-2) = %+v, want id %s", stillThere, otherID)
+ }
+}
diff --git a/apps/backend/internal/office/retention/sweep_postgres_test.go b/apps/backend/internal/office/retention/sweep_postgres_test.go
new file mode 100644
index 00000000000..06d2d62a7ff
--- /dev/null
+++ b/apps/backend/internal/office/retention/sweep_postgres_test.go
@@ -0,0 +1,238 @@
+package retention
+
+import (
+ "context"
+ "testing"
+ "time"
+
+ "github.com/jmoiron/sqlx"
+
+ "github.com/kandev/kandev/internal/db"
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+ taskrepo "github.com/kandev/kandev/internal/task/repository/sqlite"
+ "github.com/kandev/kandev/internal/testutil"
+)
+
+// newPostgresTestSweeper is sweep_test.go's newTestSweeper, built against a
+// real, isolated-schema PostgreSQL connection instead of in-memory SQLite.
+// tasks is created first, mirroring production boot order (see
+// child_summaries_postgres_test.go).
+func newPostgresTestSweeper(t *testing.T, dsn string) (*Sweeper, *sqlx.DB) {
+ t.Helper()
+ conn := testutil.OpenIsolatedPostgres(t, dsn)
+ if _, err := taskrepo.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init task repo: %v", err)
+ }
+ if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init office schema: %v", err)
+ }
+ pool := db.NewPool(conn, conn)
+ settingsRaw, err := systemsettings.NewStore(pool)
+ if err != nil {
+ t.Fatalf("init settings schema: %v", err)
+ }
+ store := NewStore(pool)
+ settingsStore := NewSettingsStore(settingsRaw)
+ previewMarker := NewPreviewMarkerStore(settingsRaw)
+ return NewSweeper(pool, store, settingsStore, previewMarker), conn
+}
+
+// TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables is
+// AC-OFFICE-RUN-HISTORY-RETENTION-002.12's mandated two-backend test: the
+// seed gives the winner two tables' worth of work, and the loser's attempt
+// happens in the pause between them (via testBetweenTablesSweep). A
+// transaction-scoped lock would have released between the winner's two
+// per-table statements and let the loser in; this must not happen with the
+// session-scoped lock.
+func TestRunSweep_TwoBackendsOnePostgres_LoserSkipsAcrossBothTables(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ winner, conn := newPostgresTestSweeper(t, dsn)
+ saveZeroFloorSettings(t, winner)
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ seedRun(t, conn, newID(), "agent-1", "finished", &old, old)
+ winner.RunSweep(ctx) // preview pass for both tables; no lock contention to test yet
+
+ loser, _ := newPostgresTestSweeper(t, dsn)
+
+ testBetweenTablesSweep = func(queryer) {
+ loser.RunSweep(ctx)
+ }
+ t.Cleanup(func() { testBetweenTablesSweep = nil })
+
+ winner.RunSweep(ctx) // deleting pass: office_routine_runs, pause, then runs
+
+ winnerLast, ok := winner.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("winner LastSweepSnapshot: ok = false, want true")
+ }
+ if winnerLast.OfficeRoutineRuns.Deleted != 1 {
+ t.Fatalf("winner office_routine_runs.Deleted = %d, want 1", winnerLast.OfficeRoutineRuns.Deleted)
+ }
+ if winnerLast.Runs.Deleted != 1 {
+ t.Fatalf("winner runs.Deleted = %d, want 1 (winner must complete both tables)", winnerLast.Runs.Deleted)
+ }
+
+ if _, ok := loser.LastSweepSnapshot(); ok {
+ t.Fatal("loser LastSweepSnapshot: ok = true, want false (loser must not have run)")
+ }
+ loserSkips, _ := loser.SkipSnapshot()
+ if loserSkips != 1 {
+ t.Fatalf("loser skip count = %d, want 1", loserSkips)
+ }
+}
+
+// TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire is
+// the companion test the design's Testing section requires: dropping the
+// winner's lock connection mid-sweep must stop it before the next table
+// (F25) and record the whole attempt as a skip rather than a partial
+// result (F28), never attempting to re-acquire.
+func TestRunSweep_LockLostMidSweep_StopsBeforeNextTableAndDoesNotReacquire(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ victim, conn := newPostgresTestSweeper(t, dsn)
+ saveZeroFloorSettings(t, victim)
+
+ // The pool's one connection (SetMaxOpenConns(1)) is what pg_terminate_backend
+ // kills below; database/sql transparently opens a replacement on the
+ // next query, which starts on the default search_path rather than
+ // this test's isolated schema. Capture the schema now to restore it
+ // before any post-mortem query on conn.
+ var schema string
+ if err := conn.Get(&schema, `SELECT current_schema()`); err != nil {
+ t.Fatalf("select current_schema: %v", err)
+ }
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ seedRun(t, conn, newID(), "agent-1", "finished", &old, old)
+ victim.RunSweep(ctx) // preview pass for both tables
+
+ admin := testutil.OpenIsolatedPostgres(t, dsn)
+ adminPool := db.NewPool(admin, admin)
+
+ testBetweenTablesSweep = func(q queryer) {
+ var pid int
+ if err := q.GetContext(ctx, &pid, `SELECT pg_backend_pid()`); err != nil {
+ t.Fatalf("select pg_backend_pid: %v", err)
+ }
+ if _, err := adminPool.Writer().ExecContext(ctx, `SELECT pg_terminate_backend($1)`, pid); err != nil {
+ t.Fatalf("terminate lock connection: %v", err)
+ }
+ // pg_terminate_backend signals the backend asynchronously; wait
+ // for it to actually leave pg_stat_activity before returning, so
+ // the alive() check right after this hook is not racing the
+ // signal's delivery.
+ deadline := time.Now().Add(5 * time.Second)
+ for {
+ var stillThere bool
+ if err := adminPool.Writer().GetContext(ctx, &stillThere,
+ `SELECT EXISTS(SELECT 1 FROM pg_stat_activity WHERE pid = $1)`, pid,
+ ); err != nil {
+ t.Fatalf("poll pg_stat_activity: %v", err)
+ }
+ if !stillThere {
+ return
+ }
+ if time.Now().After(deadline) {
+ t.Fatalf("backend %d still present in pg_stat_activity after 5s", pid)
+ }
+ time.Sleep(10 * time.Millisecond)
+ }
+ }
+ t.Cleanup(func() { testBetweenTablesSweep = nil })
+
+ beforeAttempt, ok := victim.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot after the preview pass: ok = false, want true")
+ }
+
+ victim.RunSweep(ctx) // deleting pass: office_routine_runs succeeds, then the lock connection dies
+
+ // The preview pass already set LastSweep; a lock lost mid-attempt
+ // must leave it exactly as-is rather than publishing a partial
+ // result (F28) — not become unset, which would also be true after a
+ // genuinely successful sweep with nothing yet recorded.
+ afterAttempt, ok := victim.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot after the lost-lock attempt: ok = false, want true (still the preview pass's result)")
+ }
+ if afterAttempt != beforeAttempt {
+ t.Fatalf("LastSweep changed after a lost-lock attempt: before=%+v after=%+v", beforeAttempt, afterAttempt)
+ }
+ skips, _ := victim.SkipSnapshot()
+ if skips != 1 {
+ t.Fatalf("skip count = %d, want 1", skips)
+ }
+
+ // conn's one physical connection was the one just terminated;
+ // database/sql opened a replacement on the default search_path, so
+ // restore the isolated schema before verifying table state.
+ if _, err := conn.Exec("SET search_path TO " + schema); err != nil {
+ t.Fatalf("restore search_path: %v", err)
+ }
+
+ // office_routine_runs' delete committed before the session died
+ // (AC-002.5: batches already committed stay committed).
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 {
+ t.Fatalf("office_routine_runs rows = %d, want 0 (the first table's committed delete survives)", n)
+ }
+ // runs was never reached.
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 {
+ t.Fatalf("runs rows = %d, want 1 (the second table must not have been touched)", n)
+ }
+
+ // A fresh session can now acquire the lock: it was not re-acquired by
+ // the victim, and PostgreSQL released it when the session ended.
+ replacement, ok, err := acquireSweepSession(ctx, adminPool)
+ if err != nil {
+ t.Fatalf("acquireSweepSession: %v", err)
+ }
+ if !ok {
+ t.Fatal("ok = false, want true (the terminated session must have released the lock)")
+ }
+ replacement.release()
+}
+
+// TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite
+// proves routineRunCensusTotals' window-function query — SUM(COUNT(*)) OVER
+// () over a GROUP BY routine_id — is valid PostgreSQL and, being one
+// statement, cannot be split by a write landing between the unknown-status
+// scan and the totals read, unlike the two independent queries it replaced.
+func TestCensusRoutineRuns_Postgres_TotalsQueryStaysConsistentUnderConcurrentWrite(t *testing.T) {
+ dsn := testutil.PostgresDSNFromEnv(t)
+ ctx := context.Background()
+
+ sweeper, conn := newPostgresTestSweeper(t, dsn)
+ store := sweeper.store
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutine(t, conn, "r-2")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+
+ testBetweenRoutineRunCensusReads = func(queryer) {
+ seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1))
+ seedRoutineRun(t, conn, newID(), "r-2", "done", timePtr(daysAgo(1)), daysAgo(1))
+ }
+ t.Cleanup(func() { testBetweenRoutineRunCensusReads = nil })
+
+ census, err := store.CensusRoutineRuns(ctx, conn, time.Now().UTC())
+ if err != nil {
+ t.Fatalf("CensusRoutineRuns: %v", err)
+ }
+ if census.RetainedCount != 3 {
+ t.Fatalf("retainedCount = %d, want 3 (the single totals read must see the concurrent write)", census.RetainedCount)
+ }
+ if census.TopRoutineID != "r-2" {
+ t.Fatalf("topRoutineID = %q, want r-2", census.TopRoutineID)
+ }
+ if got, want := census.TopRoutineShare, 2.0/3.0; got != want {
+ t.Fatalf("topRoutineShare = %v, want %v", got, want)
+ }
+}
diff --git a/apps/backend/internal/office/retention/sweep_test.go b/apps/backend/internal/office/retention/sweep_test.go
new file mode 100644
index 00000000000..2f4005f11a8
--- /dev/null
+++ b/apps/backend/internal/office/retention/sweep_test.go
@@ -0,0 +1,453 @@
+package retention
+
+import (
+ "context"
+ "testing"
+
+ "github.com/jmoiron/sqlx"
+ _ "github.com/mattn/go-sqlite3"
+
+ "github.com/kandev/kandev/internal/db"
+ officesqlite "github.com/kandev/kandev/internal/office/repository/sqlite"
+ systemsettings "github.com/kandev/kandev/internal/system/settings"
+)
+
+// newTestSweeper builds a Sweeper over one in-memory SQLite database
+// carrying both the office schema (routine runs, plain runs, satellites)
+// and the settings schema, so a sweep's table deletes and its
+// settings/preview reads share one connection exactly as they do on the
+// writer pool in production.
+func newTestSweeper(t *testing.T) (*Sweeper, *sqlx.DB) {
+ t.Helper()
+ conn, err := sqlx.Open("sqlite3", ":memory:?_foreign_keys=on")
+ if err != nil {
+ t.Fatalf("open sqlite: %v", err)
+ }
+ conn.SetMaxOpenConns(1)
+ t.Cleanup(func() { _ = conn.Close() })
+ if _, err := officesqlite.NewWithDB(conn, conn, nil); err != nil {
+ t.Fatalf("init office schema: %v", err)
+ }
+ pool := db.NewPool(conn, conn)
+ settingsRaw, err := systemsettings.NewStore(pool)
+ if err != nil {
+ t.Fatalf("init settings schema: %v", err)
+ }
+
+ store := NewStore(pool)
+ settingsStore := NewSettingsStore(settingsRaw)
+ previewMarker := NewPreviewMarkerStore(settingsRaw)
+ return NewSweeper(pool, store, settingsStore, previewMarker), conn
+}
+
+func TestRunSweep_FirstPassPreviewsBothTablesWithoutDeleting(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ seedRun(t, conn, newID(), "agent-1", "finished", &old, old)
+
+ sweeper.RunSweep(ctx)
+
+ last, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if !last.OfficeRoutineRuns.Previewed || last.OfficeRoutineRuns.WouldDelete != 1 {
+ t.Fatalf("office_routine_runs = %+v, want previewed with WouldDelete=1", last.OfficeRoutineRuns)
+ }
+ if last.OfficeRoutineRuns.Deleted != 0 {
+ t.Fatalf("office_routine_runs.Deleted = %d, want 0 on a preview pass", last.OfficeRoutineRuns.Deleted)
+ }
+ if !last.Runs.Previewed || last.Runs.WouldDelete != 1 {
+ t.Fatalf("runs = %+v, want previewed with WouldDelete=1", last.Runs)
+ }
+ if last.Runs.Deleted != 0 {
+ t.Fatalf("runs.Deleted = %d, want 0 on a preview pass", last.Runs.Deleted)
+ }
+
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 {
+ t.Fatalf("office_routine_runs rows after preview = %d, want 1 (nothing deleted)", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 1 {
+ t.Fatalf("runs rows after preview = %d, want 1 (nothing deleted)", n)
+ }
+}
+
+func TestRunSweep_SecondPassDeletesAfterPreview(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ runID := newID()
+ seedRun(t, conn, runID, "agent-1", "finished", &old, old)
+ seedRunEvent(t, conn, runID, 1)
+
+ sweeper.RunSweep(ctx) // preview pass
+ sweeper.RunSweep(ctx) // deleting pass
+
+ last, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if last.OfficeRoutineRuns.Previewed {
+ t.Fatal("office_routine_runs: second sweep should not be a preview pass")
+ }
+ if last.OfficeRoutineRuns.Deleted != 1 {
+ t.Fatalf("office_routine_runs.Deleted = %d, want 1", last.OfficeRoutineRuns.Deleted)
+ }
+ if last.Runs.Previewed {
+ t.Fatal("runs: second sweep should not be a preview pass")
+ }
+ if last.Runs.Deleted != 1 {
+ t.Fatalf("runs.Deleted = %d, want 1", last.Runs.Deleted)
+ }
+ if last.RunEvents.Deleted != 1 {
+ t.Fatalf("run_events.Deleted = %d, want 1", last.RunEvents.Deleted)
+ }
+
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 0 {
+ t.Fatalf("office_routine_runs rows after delete = %d, want 0", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs`); n != 0 {
+ t.Fatalf("runs rows after delete = %d, want 0", n)
+ }
+}
+
+// TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow seeds one row
+// inside the configured window and one past it, per table, with the floor
+// dropped to 0 so only age decides eligibility. A cutoff that ignores
+// WindowDays (using the sweep instant itself) would preview and then delete
+// both rows instead of only the old one.
+func TestRunSweep_RecentHistoryRowSurvivesWithinRetentionWindow(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper) // DefaultSettings() keeps the 30-day window
+
+ seedRoutine(t, conn, "r-1")
+ recent := daysAgo(1)
+ old := daysAgo(60)
+
+ recentRoutineRunID := newID()
+ oldRoutineRunID := newID()
+ seedRoutineRun(t, conn, recentRoutineRunID, "r-1", "done", &recent, recent)
+ seedRoutineRun(t, conn, oldRoutineRunID, "r-1", "done", &old, old)
+
+ recentRunID := newID()
+ oldRunID := newID()
+ seedRun(t, conn, recentRunID, "agent-1", "finished", &recent, recent)
+ seedRun(t, conn, oldRunID, "agent-1", "finished", &old, old)
+
+ sweeper.RunSweep(ctx) // preview pass
+
+ preview, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if preview.OfficeRoutineRuns.WouldDelete != 1 {
+ t.Fatalf("office_routine_runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.OfficeRoutineRuns.WouldDelete)
+ }
+ if preview.Runs.WouldDelete != 1 {
+ t.Fatalf("runs.WouldDelete = %d, want 1 (only the row past the 30-day window)", preview.Runs.WouldDelete)
+ }
+
+ sweeper.RunSweep(ctx) // deleting pass
+
+ last, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if last.OfficeRoutineRuns.Deleted != 1 {
+ t.Fatalf("office_routine_runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.OfficeRoutineRuns.Deleted)
+ }
+ if last.Runs.Deleted != 1 {
+ t.Fatalf("runs.Deleted = %d, want 1 (only the row past the 30-day window)", last.Runs.Deleted)
+ }
+
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, recentRoutineRunID); n != 1 {
+ t.Fatalf("recent routine-run rows = %d, want 1: a row inside the retention window must survive", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs WHERE id = ?`, oldRoutineRunID); n != 0 {
+ t.Fatalf("old routine-run rows = %d, want 0", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, recentRunID); n != 1 {
+ t.Fatalf("recent run rows = %d, want 1: a row inside the retention window must survive", n)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM runs WHERE id = ?`, oldRunID); n != 0 {
+ t.Fatalf("old run rows = %d, want 0", n)
+ }
+}
+
+func TestRunSweep_BacklogFlaggedWhenEligibleExceedsBatchLimit(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+
+ const eligibleRows = 105
+ const batchLimit = 100 // minBatchLimit; AC-004.3 forbids going lower
+
+ seedRoutine(t, conn, "r-1")
+ for i := 0; i < eligibleRows; i++ {
+ old := daysAgo(60 + i)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ }
+
+ settings := DefaultSettings()
+ settings.BatchLimit = batchLimit
+ settings.RoutineRuns.FloorPerOwner = 0
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ sweeper.RunSweep(ctx) // preview pass, no deletion, no batch limit involved
+ sweeper.RunSweep(ctx) // deleting pass: 105 eligible, batch limit 100
+
+ last, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false")
+ }
+ if last.OfficeRoutineRuns.Deleted != batchLimit {
+ t.Fatalf("Deleted = %d, want %d (capped by batch limit)", last.OfficeRoutineRuns.Deleted, batchLimit)
+ }
+ if !last.OfficeRoutineRuns.Backlog {
+ t.Fatal("Backlog = false, want true (105 eligible > batch limit 100)")
+ }
+}
+
+func TestRunSweep_DisabledSkipsSweepWithoutRecordingSkip(t *testing.T) {
+ sweeper, _ := newTestSweeper(t)
+ ctx := context.Background()
+
+ settings := DefaultSettings()
+ settings.Enabled = false
+ if _, err := sweeper.settingsStore.SaveSettings(ctx, settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+
+ sweeper.RunSweep(ctx)
+
+ if _, ok := sweeper.LastSweepSnapshot(); ok {
+ t.Fatal("LastSweepSnapshot: ok = true, want false (disabled means no sweep ran)")
+ }
+ count, _ := sweeper.SkipSnapshot()
+ if count != 0 {
+ t.Fatalf("skip count = %d, want 0 (disabled is not a recorded skip)", count)
+ }
+}
+
+func TestRunSweep_SettingsUnreadableSkipsAndRecordsSkip(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+
+ if _, err := conn.Exec(`
+ INSERT INTO settings (key, value, updated_at) VALUES ('office_run_retention', 'not json', CURRENT_TIMESTAMP)
+ `); err != nil {
+ t.Fatalf("seed unparseable settings: %v", err)
+ }
+
+ sweeper.RunSweep(ctx)
+
+ if _, ok := sweeper.LastSweepSnapshot(); ok {
+ t.Fatal("LastSweepSnapshot: ok = true, want false")
+ }
+ count, lastAt := sweeper.SkipSnapshot()
+ if count != 1 {
+ t.Fatalf("skip count = %d, want 1", count)
+ }
+ if lastAt.IsZero() {
+ t.Fatal("lastAt is zero, want a recorded skip time")
+ }
+}
+
+func TestRunSweep_ConcurrentAttemptRecordsSkipWithoutRunning(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+
+ // Simulate a sweep already in flight rather than racing goroutines
+ // against SQLite's speed, which would make the collision
+ // non-deterministic.
+ sweeper.mu.Lock()
+ sweeper.sweeping = true
+ sweeper.mu.Unlock()
+
+ sweeper.RunSweep(ctx)
+
+ if _, ok := sweeper.LastSweepSnapshot(); ok {
+ t.Fatal("LastSweepSnapshot: ok = true, want false (the attempt must not have run)")
+ }
+ count, _ := sweeper.SkipSnapshot()
+ if count != 1 {
+ t.Fatalf("skip count = %d, want 1", count)
+ }
+ if n := countRows(t, conn, `SELECT COUNT(*) FROM office_routine_runs`); n != 1 {
+ t.Fatalf("rows = %d, want 1 (a blocked attempt must not preview or delete)", n)
+ }
+}
+
+func TestRunSweep_AbandonedRunsBatchReportsFailureNotBacklogWithZeroSatellites(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ old := daysAgo(60)
+ runID := newID()
+ seedRun(t, conn, runID, "agent-1", "finished", &old, old)
+ seedRunEvent(t, conn, runID, 1)
+
+ sweeper.RunSweep(ctx) // preview pass
+
+ // Before each attempt's selection, make sure the row reads terminal
+ // again (undoing the previous attempt's resurrection) so it is
+ // selected every time; after selection, flip it live so that
+ // attempt's own delete re-assertion mismatches — forcing both
+ // attempts to roll back and the batch to abandon.
+ testBeforeSelectEligibleRunIDs = func(attempt int) {
+ if attempt == 0 {
+ return // already terminal from seeding
+ }
+ conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'finished', finished_at = ? WHERE id = ?`), old, runID)
+ }
+ testAfterSelectEligibleRunIDs = func(int, []string) {
+ conn.MustExec(conn.Rebind(`UPDATE runs SET status = 'queued', finished_at = NULL WHERE id = ?`), runID)
+ }
+ t.Cleanup(func() {
+ testBeforeSelectEligibleRunIDs = nil
+ testAfterSelectEligibleRunIDs = nil
+ })
+
+ sweeper.RunSweep(ctx) // deleting pass: forced to abandon
+
+ last, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false")
+ }
+ if last.Runs.Err == "" {
+ t.Fatal("runs.Err is empty, want the abandon failure recorded")
+ }
+ if last.Runs.Backlog {
+ t.Fatal("runs.Backlog = true, want false (an abandoned batch is a failure, not backlog)")
+ }
+ if last.Runs.Deleted != 0 {
+ t.Fatalf("runs.Deleted = %d, want 0", last.Runs.Deleted)
+ }
+ if last.RunEvents.Deleted != 0 {
+ t.Fatalf("run_events.Deleted = %d, want 0 (rollback restored it)", last.RunEvents.Deleted)
+ }
+}
+
+// TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState
+// is AC-OFFICE-RUN-HISTORY-RETENTION-003.4's per-table independence test:
+// office_routine_runs' preview completes successfully, but runs' own preview
+// fails in the same sweep (its table is temporarily unreachable). The next
+// sweep must not preview office_routine_runs a second time — its preview
+// already completed and a sibling's failure must not reopen it — and must
+// preview runs again, since its own preview never recorded completion.
+func TestRunSweep_SiblingTablePreviewFailureDoesNotAffectOtherTablesPreviewState(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+ saveZeroFloorSettings(t, sweeper)
+
+ seedRoutine(t, conn, "r-1")
+ old := daysAgo(60)
+ seedRoutineRun(t, conn, newID(), "r-1", "done", &old, old)
+ seedRun(t, conn, newID(), "agent-1", "finished", &old, old)
+
+ // Hide runs after office_routine_runs' own preview work has already
+ // completed for this sweep, so runs' preview fails on a genuine SQL
+ // error rather than a simulated one.
+ testBetweenTablesSweep = func(queryer) {
+ conn.MustExec(`ALTER TABLE runs RENAME TO runs_hidden`)
+ }
+ t.Cleanup(func() { testBetweenTablesSweep = nil })
+
+ sweeper.RunSweep(ctx)
+
+ first, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if !first.OfficeRoutineRuns.Previewed || first.OfficeRoutineRuns.Err != "" {
+ t.Fatalf("office_routine_runs = %+v, want a clean, completed preview", first.OfficeRoutineRuns)
+ }
+ if first.Runs.Err == "" {
+ t.Fatal("runs.Err is empty, want the preview failure recorded")
+ }
+ if first.Runs.Previewed {
+ t.Fatal("runs.Previewed = true, want false: the preview did not complete")
+ }
+
+ testBetweenTablesSweep = nil
+ conn.MustExec(`ALTER TABLE runs_hidden RENAME TO runs`)
+
+ sweeper.RunSweep(ctx)
+
+ second, ok := sweeper.LastSweepSnapshot()
+ if !ok {
+ t.Fatal("LastSweepSnapshot: ok = false, want true")
+ }
+ if second.OfficeRoutineRuns.Previewed {
+ t.Fatal("office_routine_runs.Previewed = true on the second sweep, want false: its preview already completed and must not run a second time because a sibling table failed")
+ }
+ if second.OfficeRoutineRuns.Deleted != 1 {
+ t.Fatalf("office_routine_runs.Deleted = %d, want 1 (it should now be deleting, having already completed its preview)", second.OfficeRoutineRuns.Deleted)
+ }
+ if !second.Runs.Previewed || second.Runs.Err != "" {
+ t.Fatalf("runs = %+v, want a fresh, successful preview: its earlier failed preview must not count as completed", second.Runs)
+ }
+ if second.Runs.WouldDelete != 1 {
+ t.Fatalf("runs.WouldDelete = %d, want 1", second.Runs.WouldDelete)
+ }
+}
+
+func TestLastSweepSnapshot_FalseBeforeFirstSweep(t *testing.T) {
+ sweeper, _ := newTestSweeper(t)
+ if _, ok := sweeper.LastSweepSnapshot(); ok {
+ t.Fatal("ok = true before any sweep has run, want false")
+ }
+}
+
+func TestRunCensus_PopulatesRetainedCountsForAllThreeTables(t *testing.T) {
+ sweeper, conn := newTestSweeper(t)
+ ctx := context.Background()
+
+ seedRoutine(t, conn, "r-1")
+ seedRoutineRun(t, conn, newID(), "r-1", "done", timePtr(daysAgo(1)), daysAgo(1))
+ runID := newID()
+ seedRun(t, conn, runID, "agent-1", "finished", timePtr(daysAgo(1)), daysAgo(1))
+ seedRunEvent(t, conn, runID, 1)
+
+ sweeper.RunCensus(ctx)
+
+ counts := sweeper.CensusSnapshot()
+ if counts.OfficeRoutineRuns.State != CensusFresh || counts.OfficeRoutineRuns.RetainedCount != 1 {
+ t.Fatalf("office_routine_runs census = %+v, want fresh with 1", counts.OfficeRoutineRuns)
+ }
+ if counts.Runs.State != CensusFresh || counts.Runs.RetainedCount != 1 {
+ t.Fatalf("runs census = %+v, want fresh with 1", counts.Runs)
+ }
+ if counts.RunEvents.State != CensusFresh || counts.RunEvents.RetainedCount != 1 {
+ t.Fatalf("run_events census = %+v, want fresh with 1", counts.RunEvents)
+ }
+}
+
+// saveZeroFloorSettings drops both tables' floor to 0 so a test's single
+// seeded row is eligible: the default floor of 50 protects the newest 50
+// rows per owner, which a one-row fixture never exceeds.
+func saveZeroFloorSettings(t *testing.T, sweeper *Sweeper) {
+ t.Helper()
+ settings := DefaultSettings()
+ settings.RoutineRuns.FloorPerOwner = 0
+ settings.Runs.FloorPerOwner = 0
+ if _, err := sweeper.settingsStore.SaveSettings(context.Background(), settings); err != nil {
+ t.Fatalf("SaveSettings: %v", err)
+ }
+}
diff --git a/apps/backend/internal/office/retention/types.go b/apps/backend/internal/office/retention/types.go
new file mode 100644
index 00000000000..63e7451353b
--- /dev/null
+++ b/apps/backend/internal/office/retention/types.go
@@ -0,0 +1,146 @@
+// Package retention bounds Office run history: office_routine_runs, runs,
+// and the run satellite tables (run_events, office_run_route_attempts,
+// office_run_skills). See docs/specs/office/requirements/run-history-retention.md
+// and its operations counterpart for the frozen contract this package
+// implements.
+package retention
+
+import (
+ "errors"
+ "fmt"
+)
+
+// ErrValidation is returned by NormalizeSettings when a field is outside its
+// permitted range. The error message names the field.
+var ErrValidation = errors.New("retention settings validation")
+
+// ErrInvalidPersistedSettings wraps a stored settings document that could
+// not be read or parsed. Callers fall back to DefaultSettings and surface a
+// health issue rather than failing.
+var ErrInvalidPersistedSettings = errors.New("invalid persisted retention settings")
+
+// TableName identifies one of the tables retention reasons about.
+type TableName string
+
+const (
+ TableOfficeRoutineRuns TableName = "office_routine_runs"
+ TableRuns TableName = "runs"
+ TableRunEvents TableName = "run_events"
+)
+
+// SweptTables are the two tables retention selects rows from by policy, in
+// the fixed sweep order.
+var SweptTables = []TableName{TableOfficeRoutineRuns, TableRuns}
+
+// ReportedTables are the swept tables plus the three run satellites that a
+// sweep can delete rows from.
+var ReportedTables = []TableName{
+ TableOfficeRoutineRuns, TableRuns,
+ "run_events", "office_run_route_attempts", "office_run_skills",
+}
+
+// ThresholdedTables carry a warning threshold on retained row count.
+var ThresholdedTables = []TableName{TableOfficeRoutineRuns, TableRuns, TableRunEvents}
+
+// TableSettings is the per-table retention policy for a swept table.
+type TableSettings struct {
+ WindowDays int `json:"window_days"`
+ FloorPerOwner int `json:"floor_per_owner"`
+ WarnRows int `json:"warn_rows"`
+}
+
+// RunEventsSettings is the threshold-only policy for run_events, which has
+// no window or floor of its own: its lifetime is its run's.
+type RunEventsSettings struct {
+ WarnRows int `json:"warn_rows"`
+}
+
+// Settings is the full retention policy document, persisted under one
+// settings-store key as JSON.
+type Settings struct {
+ Enabled bool `json:"enabled"`
+ SweepIntervalHours int `json:"sweep_interval_hours"`
+ BatchLimit int `json:"batch_limit"`
+ RoutineRuns TableSettings `json:"routine_runs"`
+ Runs TableSettings `json:"runs"`
+ RunEvents RunEventsSettings `json:"run_events"`
+}
+
+// DefaultSettings returns the documented AC-OFFICE-RUN-HISTORY-RETENTION-004.2
+// defaults.
+func DefaultSettings() Settings {
+ return Settings{
+ Enabled: true,
+ SweepIntervalHours: 6,
+ BatchLimit: 5000,
+ RoutineRuns: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000},
+ Runs: TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000},
+ RunEvents: RunEventsSettings{WarnRows: 250000},
+ }
+}
+
+// Permitted ranges, AC-OFFICE-RUN-HISTORY-RETENTION-004.3.
+const (
+ minWindowDays = 1
+ maxWindowDays = 3650
+
+ minSweepIntervalHours = 1
+ maxSweepIntervalHours = 168
+
+ minFloorPerOwner = 0
+ maxFloorPerOwner = 10000
+
+ minBatchLimit = 100
+ maxBatchLimit = 100000
+
+ minWarnRows = 0
+)
+
+// NormalizeSettings validates every field against its permitted range and
+// returns a field-named error on the first violation, changing nothing.
+// Ranges are inclusive on both ends.
+func NormalizeSettings(in Settings) (Settings, error) {
+ if err := validateRange("sweep_interval_hours", in.SweepIntervalHours, minSweepIntervalHours, maxSweepIntervalHours); err != nil {
+ return Settings{}, err
+ }
+ if err := validateRange("batch_limit", in.BatchLimit, minBatchLimit, maxBatchLimit); err != nil {
+ return Settings{}, err
+ }
+ if err := validateTableSettings("routine_runs", in.RoutineRuns); err != nil {
+ return Settings{}, err
+ }
+ if err := validateTableSettings("runs", in.Runs); err != nil {
+ return Settings{}, err
+ }
+ if err := validateMin("run_events.warn_rows", in.RunEvents.WarnRows, minWarnRows); err != nil {
+ return Settings{}, err
+ }
+ return in, nil
+}
+
+func validateTableSettings(prefix string, s TableSettings) error {
+ if err := validateRange(prefix+".window_days", s.WindowDays, minWindowDays, maxWindowDays); err != nil {
+ return err
+ }
+ if err := validateRange(prefix+".floor_per_owner", s.FloorPerOwner, minFloorPerOwner, maxFloorPerOwner); err != nil {
+ return err
+ }
+ if err := validateMin(prefix+".warn_rows", s.WarnRows, minWarnRows); err != nil {
+ return err
+ }
+ return nil
+}
+
+func validateRange(field string, value, minValue, maxValue int) error {
+ if value < minValue || value > maxValue {
+ return fmt.Errorf("%w: %s must be between %d and %d", ErrValidation, field, minValue, maxValue)
+ }
+ return nil
+}
+
+func validateMin(field string, value, minValue int) error {
+ if value < minValue {
+ return fmt.Errorf("%w: %s must be %d or greater", ErrValidation, field, minValue)
+ }
+ return nil
+}
diff --git a/apps/backend/internal/office/retention/types_test.go b/apps/backend/internal/office/retention/types_test.go
new file mode 100644
index 00000000000..00b61d94d14
--- /dev/null
+++ b/apps/backend/internal/office/retention/types_test.go
@@ -0,0 +1,94 @@
+package retention
+
+import "testing"
+
+func TestDefaultSettings_MatchesDocumentedDefaults(t *testing.T) {
+ got := DefaultSettings()
+
+ if !got.Enabled {
+ t.Fatalf("Enabled = false, want true")
+ }
+ if got.SweepIntervalHours != 6 {
+ t.Fatalf("SweepIntervalHours = %d, want 6", got.SweepIntervalHours)
+ }
+ if got.BatchLimit != 5000 {
+ t.Fatalf("BatchLimit = %d, want 5000", got.BatchLimit)
+ }
+ wantRoutineRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}
+ if got.RoutineRuns != wantRoutineRuns {
+ t.Fatalf("RoutineRuns = %+v, want %+v", got.RoutineRuns, wantRoutineRuns)
+ }
+ wantRuns := TableSettings{WindowDays: 30, FloorPerOwner: 50, WarnRows: 25000}
+ if got.Runs != wantRuns {
+ t.Fatalf("Runs = %+v, want %+v", got.Runs, wantRuns)
+ }
+ if got.RunEvents.WarnRows != 250000 {
+ t.Fatalf("RunEvents.WarnRows = %d, want 250000", got.RunEvents.WarnRows)
+ }
+}
+
+func TestNormalizeSettings_AcceptsDefaults(t *testing.T) {
+ normalized, err := NormalizeSettings(DefaultSettings())
+ if err != nil {
+ t.Fatalf("NormalizeSettings(defaults): %v", err)
+ }
+ if normalized != DefaultSettings() {
+ t.Fatalf("NormalizeSettings(defaults) = %+v, want unchanged defaults", normalized)
+ }
+}
+
+func TestNormalizeSettings_RejectsOutOfRangeFields(t *testing.T) {
+ cases := []struct {
+ name string
+ mutate func(*Settings)
+ wantErr string
+ }{
+ {"window too low", func(s *Settings) { s.RoutineRuns.WindowDays = 0 }, "routine_runs.window_days"},
+ {"window too high", func(s *Settings) { s.Runs.WindowDays = 3651 }, "runs.window_days"},
+ {"interval too low", func(s *Settings) { s.SweepIntervalHours = 0 }, "sweep_interval_hours"},
+ {"interval too high", func(s *Settings) { s.SweepIntervalHours = 169 }, "sweep_interval_hours"},
+ {"floor negative", func(s *Settings) { s.RoutineRuns.FloorPerOwner = -1 }, "routine_runs.floor_per_owner"},
+ {"floor too high", func(s *Settings) { s.Runs.FloorPerOwner = 10001 }, "runs.floor_per_owner"},
+ {"batch too low", func(s *Settings) { s.BatchLimit = 99 }, "batch_limit"},
+ {"batch too high", func(s *Settings) { s.BatchLimit = 100001 }, "batch_limit"},
+ {"warn negative", func(s *Settings) { s.RoutineRuns.WarnRows = -1 }, "routine_runs.warn_rows"},
+ {"run_events warn negative", func(s *Settings) { s.RunEvents.WarnRows = -1 }, "run_events.warn_rows"},
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ s := DefaultSettings()
+ tc.mutate(&s)
+ _, err := NormalizeSettings(s)
+ if err == nil {
+ t.Fatalf("NormalizeSettings(%+v): want error, got nil", s)
+ }
+ if got := err.Error(); !containsField(got, tc.wantErr) {
+ t.Fatalf("NormalizeSettings error = %q, want it to name field %q", got, tc.wantErr)
+ }
+ })
+ }
+}
+
+func TestNormalizeSettings_ZeroWarnRowsDisablesThreshold(t *testing.T) {
+ s := DefaultSettings()
+ s.RoutineRuns.WarnRows = 0
+ s.Runs.WarnRows = 0
+ s.RunEvents.WarnRows = 0
+ if _, err := NormalizeSettings(s); err != nil {
+ t.Fatalf("NormalizeSettings with zero warn thresholds: %v", err)
+ }
+}
+
+func containsField(msg, field string) bool {
+ return len(msg) >= len(field) && (indexOf(msg, field) >= 0)
+}
+
+func indexOf(haystack, needle string) int {
+ for i := 0; i+len(needle) <= len(haystack); i++ {
+ if haystack[i:i+len(needle)] == needle {
+ return i
+ }
+ }
+ return -1
+}
diff --git a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go
index a77e43d3c18..7952730f079 100644
--- a/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go
+++ b/apps/backend/internal/office/service/event_subscribers_lost_race_side_effects_test.go
@@ -47,9 +47,17 @@ func TestHandleAgentCompleted_FinishRunFailureSkipsCompletionSideEffects(t *test
// Force FinishRun's UPDATE to fail with a genuine error, matching
// TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails'
- // fault injection: a targeted fault (drop the column FinishRun sets)
- // rather than a global read-only pragma.
- svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at")
+ // targeted fault rather than a global read-only pragma.
+ // Block only the terminal timestamp update. The retention expression index
+ // references finished_at, so dropping the column would fail before the
+ // handler runs and would no longer exercise its guarded error path.
+ svc.ExecSQL(t, `
+ CREATE TRIGGER block_completion_finish_test
+ BEFORE UPDATE OF finished_at ON runs
+ WHEN NEW.finished_at IS NOT NULL
+ BEGIN
+ SELECT RAISE(FAIL, 'finished_at update blocked for test');
+ END`)
completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{
"task_id": taskID,
@@ -114,7 +122,13 @@ func TestHandleTasklessAgentCompleted_FinishRunFailureSkipsCompletionSideEffects
)
`, agent.ID)
- svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at")
+ svc.ExecSQL(t, `
+ CREATE TRIGGER block_taskless_completion_finish_test
+ BEFORE UPDATE OF finished_at ON runs
+ WHEN NEW.finished_at IS NOT NULL
+ BEGIN
+ SELECT RAISE(FAIL, 'finished_at update blocked for test');
+ END`)
completed := bus.NewEvent(events.AgentCompleted, "test", map[string]string{
"agent_id": agent.ID,
diff --git a/apps/backend/internal/office/service/scheduler_checkout_error_test.go b/apps/backend/internal/office/service/scheduler_checkout_error_test.go
index 1cd259ec9ee..05f250781ac 100644
--- a/apps/backend/internal/office/service/scheduler_checkout_error_test.go
+++ b/apps/backend/internal/office/service/scheduler_checkout_error_test.go
@@ -120,8 +120,17 @@ func TestSchedulerTick_AgentCompletedKeepsCheckoutWhenFinishRunFails(t *testing.
// the tasks table, and every other runs column, writable — a targeted
// fault instead of a global read-only pragma, so this test actually
// distinguishes "release before finish" from "finish before release"
- // rather than failing both writes identically.
- svc.ExecSQL(t, "ALTER TABLE runs DROP COLUMN finished_at")
+ // rather than failing both writes identically. A trigger rather than
+ // DROP COLUMN: idx_runs_retention is an expression index over
+ // COALESCE(finished_at, ...), and SQLite refuses to drop a column an
+ // index still references.
+ svc.ExecSQL(t, `
+ CREATE TRIGGER block_finish_order_test
+ BEFORE UPDATE OF finished_at ON runs
+ WHEN NEW.finished_at IS NOT NULL
+ BEGIN
+ SELECT RAISE(FAIL, 'finished_at update blocked for test');
+ END`)
event := bus.NewEvent(events.AgentCompleted, "test", map[string]string{
"task_id": "task-finish-order-1",
diff --git a/apps/web/components/settings/system/data-logs-settings.tsx b/apps/web/components/settings/system/data-logs-settings.tsx
index 93ff76a4c46..5ef25f4601a 100644
--- a/apps/web/components/settings/system/data-logs-settings.tsx
+++ b/apps/web/components/settings/system/data-logs-settings.tsx
@@ -7,6 +7,7 @@ import { SettingsTarget } from "@/components/settings/settings-target";
import { BackupsTable } from "@/components/settings/system/backups-table";
import { DatabaseStatsCard } from "@/components/settings/system/database-stats-card";
import { LogViewer } from "@/components/settings/system/log-viewer";
+import { RetentionSettingsCard } from "@/components/settings/system/retention-settings-card";
import { BACKUP_SQL_COMMAND } from "@/components/settings/system/system-route-shell";
import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system";
@@ -35,6 +36,14 @@ export function DataLogsSettings() {
+
+
+
+
+ ({
+ fetchRetentionStatus: (...args: unknown[]) => fetchRetentionStatusMock(...args),
+ saveRetentionSettings: (...args: unknown[]) => saveRetentionSettingsMock(...args),
+}));
+
+vi.mock("@/components/settings/settings-save-provider", () => ({
+ useSettingsSaveContributor: (contributor: SettingsSaveContributor) => {
+ saveContributor = contributor;
+ },
+}));
+
+import { RetentionSettingsCard } from "./retention-settings-card";
+
+function defaultSettings(overrides: Partial = {}): RetentionSettings {
+ return {
+ enabled: true,
+ sweep_interval_hours: 6,
+ batch_limit: 5000,
+ routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ run_events: { warn_rows: 250000 },
+ ...overrides,
+ };
+}
+
+function statusOf(overrides: Partial = {}): RetentionStatus {
+ return {
+ settings: defaultSettings(),
+ last_sweep: null,
+ skip_count: 0,
+ retained_counts: {
+ office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" },
+ runs: { state: "not_computed", retained_count: 0, as_of: "" },
+ run_events: { state: "not_computed", retained_count: 0, as_of: "" },
+ },
+ ...overrides,
+ };
+}
+
+function renderCard() {
+ return render(
+
+
+ ,
+ );
+}
+
+beforeEach(() => {
+ fetchRetentionStatusMock.mockReset();
+ saveRetentionSettingsMock.mockReset();
+ fetchRetentionStatusMock.mockResolvedValue(statusOf());
+ currentRole = "admin";
+ saveContributor = null;
+});
+
+afterEach(() => {
+ cleanup();
+ vi.clearAllMocks();
+});
+
+describe("RetentionSettingsCard", () => {
+ it("loads settings and renders the swept-table fields from the fetched status", async () => {
+ renderCard();
+
+ const windowDays = await screen.findByTestId("retention-routine-runs-window-days");
+ expect(windowDays).toHaveProperty("value", "30");
+ expect(screen.getByTestId("retention-runs-warn-rows")).toHaveProperty("value", "25000");
+ expect(screen.getByTestId("retention-run-events-warn-rows")).toHaveProperty("value", "250000");
+ expect(screen.getByTestId("retention-never-swept")).toBeTruthy();
+ });
+
+ it("keeps members read-only while preserving the loaded values", async () => {
+ currentRole = "member";
+ renderCard();
+
+ const windowDays = await screen.findByTestId("retention-routine-runs-window-days");
+ expect(windowDays).toHaveProperty("disabled", true);
+ expect(screen.getByTestId(ENABLED_TOGGLE_TEST_ID)).toHaveProperty("disabled", true);
+ expect(screen.getByText("Only an admin can change retention settings.")).toBeTruthy();
+ expect(saveContributor?.isDirty).toBe(false);
+ });
+
+ it("reports a failed save without clearing the dirty draft", async () => {
+ renderCard();
+ await screen.findByTestId(ENABLED_TOGGLE_TEST_ID);
+ fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID));
+ if (!saveContributor) throw new Error("expected save contributor");
+
+ saveRetentionSettingsMock.mockRejectedValueOnce(new Error("offline"));
+ await act(async () => {
+ await expect(saveContributor?.save(saveContributor.revision)).rejects.toThrow("offline");
+ });
+
+ expect(saveContributor?.isDirty).toBe(true);
+ await screen.findByTestId("retention-save-error");
+ });
+
+ it("renders the last sweep outcome, backlog flag, and retained counts", async () => {
+ fetchRetentionStatusMock.mockResolvedValue(
+ statusOf({
+ last_sweep: {
+ started_at: "2026-09-01T00:00:00Z",
+ finished_at: "2026-09-01T00:00:05Z",
+ office_routine_runs: {
+ deleted: 12,
+ backlog: true,
+ error: "",
+ previewed: false,
+ would_delete: 0,
+ },
+ runs: { deleted: 3, backlog: false, error: "", previewed: true, would_delete: 40 },
+ run_events: { deleted: 100, backlog: false, error: "" },
+ route_attempts: { deleted: 0, backlog: false, error: "" },
+ run_skills: { deleted: 0, backlog: false, error: "" },
+ },
+ skip_count: 2,
+ last_skip_at: "2026-09-01T00:10:00Z",
+ retained_counts: {
+ office_routine_runs: {
+ state: "fresh",
+ retained_count: 1200,
+ as_of: "2026-09-01T00:00:00Z",
+ top_routine_id: "routine-1",
+ top_routine_share: 0.42,
+ },
+ runs: { state: "stale", retained_count: 800, as_of: "2026-08-31T00:00:00Z" },
+ run_events: { state: "not_computed", retained_count: 0, as_of: "" },
+ },
+ }),
+ );
+ renderCard();
+
+ await screen.findByTestId("retention-last-sweep");
+ expect(screen.getByTestId("retention-backlog-office_routine_runs")).toBeTruthy();
+ expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain(
+ "1200",
+ );
+ expect(screen.getByTestId("retention-retained-office_routine_runs").textContent).toContain(
+ "42%",
+ );
+ expect(screen.getByText(/Stale: last measurement failed/)).toBeTruthy();
+ expect(screen.getByTestId("retention-skip-count").textContent).toContain("2");
+ });
+});
+
+describe("RetentionSettingsCard save/reload consistency", () => {
+ it("stages an admin edit until the shared save contributor runs, then reloads", async () => {
+ renderCard();
+ await screen.findByTestId(ENABLED_TOGGLE_TEST_ID);
+
+ const toggle = screen.getByTestId(ENABLED_TOGGLE_TEST_ID);
+ fireEvent.click(toggle);
+ expect(saveRetentionSettingsMock).not.toHaveBeenCalled();
+ expect(saveContributor?.isDirty).toBe(true);
+ if (!saveContributor) throw new Error("expected save contributor");
+
+ saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false }));
+ fetchRetentionStatusMock.mockResolvedValueOnce(
+ statusOf({ settings: defaultSettings({ enabled: false }) }),
+ );
+
+ await act(async () => saveContributor?.save(saveContributor.revision));
+
+ expect(saveRetentionSettingsMock).toHaveBeenCalledWith(
+ expect.objectContaining({ enabled: false }),
+ );
+ await waitFor(() => expect(saveContributor?.isDirty).toBe(false));
+ });
+
+ it("clears the dirty draft from the save response even when the post-save reload fails", async () => {
+ renderCard();
+ await screen.findByTestId(ENABLED_TOGGLE_TEST_ID);
+ fireEvent.click(screen.getByTestId(ENABLED_TOGGLE_TEST_ID));
+ if (!saveContributor) throw new Error("expected save contributor");
+
+ saveRetentionSettingsMock.mockResolvedValueOnce(defaultSettings({ enabled: false }));
+ fetchRetentionStatusMock.mockRejectedValueOnce(new Error("offline"));
+
+ await act(async () => saveContributor?.save(saveContributor.revision));
+
+ expect(saveRetentionSettingsMock).toHaveBeenCalledWith(
+ expect.objectContaining({ enabled: false }),
+ );
+ await waitFor(() => expect(saveContributor?.isDirty).toBe(false));
+ });
+});
+
+describe("RetentionSettingsCard unknown-status reporting", () => {
+ it("renders each unrecognized status with its row count", async () => {
+ fetchRetentionStatusMock.mockResolvedValue(
+ statusOf({
+ retained_counts: {
+ office_routine_runs: {
+ state: "fresh",
+ retained_count: 5,
+ as_of: "2026-09-01T00:00:00Z",
+ unknown_statuses: [{ status: "quarantined", count: 2 }],
+ },
+ runs: { state: "not_computed", retained_count: 0, as_of: "" },
+ run_events: { state: "not_computed", retained_count: 0, as_of: "" },
+ },
+ }),
+ );
+ renderCard();
+
+ const row = await screen.findByTestId("retention-retained-office_routine_runs");
+ expect(row.textContent).toContain("quarantined (2)");
+ });
+});
diff --git a/apps/web/components/settings/system/retention-settings-card.tsx b/apps/web/components/settings/system/retention-settings-card.tsx
new file mode 100644
index 00000000000..c08b97fb35f
--- /dev/null
+++ b/apps/web/components/settings/system/retention-settings-card.tsx
@@ -0,0 +1,557 @@
+"use client";
+
+import { useEffect, useRef, useState, type ReactNode } from "react";
+import { useTranslation } from "react-i18next";
+import { Alert, AlertDescription } from "@kandev/ui/alert";
+import { CardContent } from "@kandev/ui/card";
+import { Input } from "@kandev/ui/input";
+import { Spinner } from "@kandev/ui/spinner";
+import { Switch } from "@kandev/ui/switch";
+import { IconAlertCircle } from "@tabler/icons-react";
+import { SettingsCard } from "@/components/settings/settings-card";
+import { SettingsCardHeader } from "@/components/settings/settings-card-header";
+import { settingsControlClassName } from "@/components/settings/settings-control";
+import {
+ SettingsFieldDescription,
+ SettingsFieldLabel,
+} from "@/components/settings/settings-typography";
+import { useSettingsSaveContributor } from "@/components/settings/settings-save-provider";
+import { useIsAdmin } from "@/hooks/domains/auth/use-is-admin";
+import { useRetentionSettings } from "@/hooks/domains/system/use-retention-settings";
+import { formatDateTime } from "@/lib/i18n/formats";
+import { SYSTEM_SETTINGS_TARGETS } from "@/lib/settings-discovery/catalog/system";
+import type {
+ RetentionSettings,
+ RetentionStatus,
+ RetentionTableCensus,
+ RetentionTableSweepResult,
+ RetentionSweptTableResult,
+ RetentionUnknownStatusCount,
+} from "@/lib/types/system";
+
+function serialize(settings: RetentionSettings | null): string {
+ return settings ? JSON.stringify(settings) : "loading";
+}
+
+function formatUnknownStatuses(unknown: RetentionUnknownStatusCount[]): string {
+ return unknown.map((u) => `${u.status} (${u.count})`).join(", ");
+}
+
+function NumberField({
+ label,
+ help,
+ value,
+ min,
+ max,
+ disabled,
+ onChange,
+ testId,
+}: {
+ label: string;
+ help: string;
+ value: number;
+ min: number;
+ max?: number;
+ disabled?: boolean;
+ onChange: (value: number) => void;
+ testId: string;
+}) {
+ return (
+
+ );
+}
diff --git a/apps/web/components/settings/system/system-route-copy.test.ts b/apps/web/components/settings/system/system-route-copy.test.ts
index 669405ac22c..1024b5e15df 100644
--- a/apps/web/components/settings/system/system-route-copy.test.ts
+++ b/apps/web/components/settings/system/system-route-copy.test.ts
@@ -17,6 +17,7 @@ vi.mock("@/components/settings/settings-target", () => ({
vi.mock("./backups-table", () => ({ BackupsTable: () => null }));
vi.mock("./database-stats-card", () => ({ DatabaseStatsCard: () => null }));
vi.mock("./log-viewer", () => ({ LogViewer: () => null }));
+vi.mock("./retention-settings-card", () => ({ RetentionSettingsCard: () => null }));
afterEach(() => {
cleanup();
databaseState.value = null;
@@ -147,6 +148,7 @@ describe("Data & Logs composition", () => {
render(createElement(DataLogsSettings));
expect(screen.getByText(t("system:navDatabase"))).toBeTruthy();
+ expect(screen.getByText(t("system:navRetention"))).toBeTruthy();
expect(screen.getByText(t("system:navBackups"))).toBeTruthy();
expect(screen.getByText(t("system:navLogs"))).toBeTruthy();
expect(screen.queryByText(t("system:storageTitle"))).toBeNull();
diff --git a/apps/web/e2e/tests/system/retention-settings.spec.ts b/apps/web/e2e/tests/system/retention-settings.spec.ts
new file mode 100644
index 00000000000..08a75c77394
--- /dev/null
+++ b/apps/web/e2e/tests/system/retention-settings.spec.ts
@@ -0,0 +1,97 @@
+import { test, expect } from "../../fixtures/test-base";
+
+/**
+ * Office run-history retention (docs/specs/office/requirements/run-history-retention*.md).
+ * The sweep itself is time- and seed-dependent and is covered far more cheaply by backend
+ * tests; the two honest E2E candidates are the settings round-trip through the real API
+ * and a threshold warning actually reaching the Health card. Both tests restore whatever
+ * global retention settings they change, since retention settings are process-global and
+ * this worker's backend is reused by every test file in the shard.
+ */
+test.describe("System retention settings", () => {
+ test("persists an admin policy edit through save and across a backend restart", async ({
+ testPage,
+ backend,
+ }) => {
+ test.setTimeout(90_000);
+ await testPage.goto("/settings/system/data-storage");
+ const batchLimit = testPage.getByTestId("retention-batch-limit");
+ await expect(batchLimit).toBeVisible();
+ const original = await batchLimit.inputValue();
+ const updated = original === "7777" ? "8888" : "7777";
+
+ try {
+ await batchLimit.fill(updated);
+ await testPage.getByRole("button", { name: "Save changes" }).click();
+ await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved");
+
+ await testPage.reload();
+ await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated);
+
+ await backend.restart();
+ await testPage.reload();
+ await expect(testPage.getByTestId("retention-batch-limit")).toHaveValue(updated);
+ } finally {
+ await testPage.getByTestId("retention-batch-limit").fill(original);
+ await testPage.getByRole("button", { name: "Save changes" }).click();
+ await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved");
+ }
+ });
+
+ test("shows a threshold warning on the Health card once retained runs exceed the configured limit", async ({
+ testPage,
+ backend,
+ apiClient,
+ seedData,
+ }) => {
+ test.setTimeout(90_000);
+ await testPage.goto("/settings/system/data-storage");
+ const warnField = testPage.getByTestId("retention-runs-warn-rows");
+ await expect(warnField).toBeVisible();
+ const originalWarnRows = await warnField.inputValue();
+
+ const initialStatus = await testPage.evaluate(async () => {
+ const response = await fetch("/api/v1/system/retention");
+ return response.json();
+ });
+ const baselineRetained =
+ initialStatus.retained_counts.runs.state === "fresh"
+ ? initialStatus.retained_counts.runs.retained_count
+ : 0;
+ // warn_rows=0 disables the threshold check entirely (AC-OFFICE-RUN-HISTORY-RETENTION-004.3),
+ // and seeding 3 new terminal runs guarantees the post-seed count clears baseline+1 even when
+ // baseline is 0.
+ const seededRunCount = 3;
+ const newWarnRows = baselineRetained + 1;
+ const expectedRetained = baselineRetained + seededRunCount;
+ for (let i = 0; i < seededRunCount; i++) {
+ await apiClient.seedRun({ agentProfileId: seedData.agentProfileId, status: "finished" });
+ }
+
+ try {
+ await warnField.fill(String(newWarnRows));
+ await testPage.getByRole("button", { name: "Save changes" }).click();
+ await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved");
+
+ // The census only re-evaluates on its interval timer or at scheduler Start; a
+ // restart forces an immediate re-evaluation against the settings just saved and
+ // the runs just seeded, deterministically, without waiting out the real interval.
+ await backend.restart();
+
+ await testPage.goto("/settings/system/status");
+ const issue = testPage.getByTestId("system-health-issue-office_retention_threshold:runs");
+ await expect(issue).toBeVisible({ timeout: 15_000 });
+ await expect(issue).toContainText("over its threshold");
+
+ await testPage.goto("/settings/system/data-storage");
+ const retainedRuns = testPage.getByTestId("retention-retained-runs");
+ await expect(retainedRuns).toBeVisible();
+ await expect(retainedRuns).toContainText(String(expectedRetained));
+ } finally {
+ await testPage.goto("/settings/system/data-storage");
+ await testPage.getByTestId("retention-runs-warn-rows").fill(originalWarnRows);
+ await testPage.getByRole("button", { name: "Save changes" }).click();
+ await expect(testPage.getByTestId("settings-floating-save")).toContainText("Saved");
+ }
+ });
+});
diff --git a/apps/web/hooks/domains/system/use-retention-settings.ts b/apps/web/hooks/domains/system/use-retention-settings.ts
new file mode 100644
index 00000000000..d93f2d6deb6
--- /dev/null
+++ b/apps/web/hooks/domains/system/use-retention-settings.ts
@@ -0,0 +1,67 @@
+"use client";
+
+import { useCallback, useEffect, useState } from "react";
+import { useAppStore, useAppStoreApi } from "@/components/state-provider";
+import { fetchRetentionStatus, saveRetentionSettings } from "@/lib/api/domains/system-api";
+import type { RetentionSettings } from "@/lib/types/system";
+
+export function useRetentionSettings() {
+ const status = useAppStore((s) => s.system.retention);
+ const setStatus = useAppStore((s) => s.setSystemRetention);
+ const storeApi = useAppStoreApi();
+ const [isLoading, setIsLoading] = useState(false);
+ const [error, setError] = useState(null);
+ const [saveError, setSaveError] = useState(null);
+
+ const reload = useCallback(async () => {
+ setIsLoading(true);
+ setError(null);
+ try {
+ setStatus(await fetchRetentionStatus({ cache: "no-store" }));
+ } catch (e) {
+ setError(e instanceof Error ? e.message : String(e));
+ } finally {
+ setIsLoading(false);
+ }
+ }, [setStatus]);
+
+ useEffect(() => {
+ if (status) return;
+ void reload();
+ }, [status, reload]);
+
+ const save = useCallback(
+ async (settings: RetentionSettings) => {
+ setSaveError(null);
+ try {
+ const saved = await saveRetentionSettings(settings);
+ // Apply the PUT's own normalized response synchronously, rather
+ // than relying solely on the reload() below: if that GET fails,
+ // its error is recorded but never surfaces once status is already
+ // loaded (see the isLoading/error-gated branches in
+ // RetentionSettingsCard), which would otherwise leave the store
+ // holding pre-save settings while the save coordinator believes
+ // the save already succeeded.
+ storeApi.setState((state) =>
+ state.system.retention
+ ? {
+ system: {
+ ...state.system,
+ retention: { ...state.system.retention, settings: saved },
+ },
+ }
+ : state,
+ );
+ void reload();
+ return saved;
+ } catch (e) {
+ const message = e instanceof Error ? e.message : String(e);
+ setSaveError(message);
+ throw e;
+ }
+ },
+ [reload, storeApi],
+ );
+
+ return { status, isLoading, error, saveError, reload, save };
+}
diff --git a/apps/web/lib/api/domains/system-api.test.ts b/apps/web/lib/api/domains/system-api.test.ts
index c1980b1dd63..fda9156ac2c 100644
--- a/apps/web/lib/api/domains/system-api.test.ts
+++ b/apps/web/lib/api/domains/system-api.test.ts
@@ -43,6 +43,8 @@ import {
restoreStorageQuarantine,
runStorageMaintenance,
saveStorageSettings,
+ fetchRetentionStatus,
+ saveRetentionSettings,
} from "./system-api";
const BASE = "http://api.test/api/v1/system";
@@ -528,3 +530,47 @@ describe("storage policy", () => {
expect(response.capabilities.docker_available).toBe(true);
});
});
+
+describe("office run history retention", () => {
+ const retentionSettings = {
+ enabled: true,
+ sweep_interval_hours: 6,
+ batch_limit: 5000,
+ routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ run_events: { warn_rows: 250000 },
+ };
+
+ it("loads retention status without caching", async () => {
+ fetchSpy.mockResolvedValueOnce(
+ jsonResponse({
+ settings: retentionSettings,
+ last_sweep: null,
+ skip_count: 0,
+ retained_counts: {
+ office_routine_runs: { state: "not_computed", retained_count: 0, as_of: "" },
+ runs: { state: "not_computed", retained_count: 0, as_of: "" },
+ run_events: { state: "not_computed", retained_count: 0, as_of: "" },
+ },
+ }),
+ );
+
+ const response = await fetchRetentionStatus();
+
+ expect(lastCall().url).toBe(`${BASE}/retention`);
+ expect(lastCall().init?.cache).toBe("no-store");
+ expect(response.settings).toEqual(retentionSettings);
+ expect(response.last_sweep).toBeNull();
+ });
+
+ it("PUTs the full settings document to save", async () => {
+ fetchSpy.mockResolvedValueOnce(jsonResponse(retentionSettings));
+
+ const response = await saveRetentionSettings(retentionSettings);
+
+ expect(lastCall().url).toBe(`${BASE}/retention`);
+ expect(method()).toBe("PUT");
+ expect(JSON.parse(String(lastCall().init?.body))).toEqual(retentionSettings);
+ expect(response).toEqual(retentionSettings);
+ });
+});
diff --git a/apps/web/lib/api/domains/system-api.ts b/apps/web/lib/api/domains/system-api.ts
index 8c27004e7b8..4c5395cfd27 100644
--- a/apps/web/lib/api/domains/system-api.ts
+++ b/apps/web/lib/api/domains/system-api.ts
@@ -24,6 +24,8 @@ import type {
StorageQuarantinePurgeScope,
StorageSettingsResponse,
UpdatesChannel,
+ RetentionSettings,
+ RetentionStatus,
} from "@/lib/types/system";
const SYSTEM_BASE = "/api/v1/system";
@@ -416,3 +418,26 @@ export function purgeStorageQuarantine(
},
});
}
+
+// --- Office run history retention ----------------------------------------
+
+export function fetchRetentionStatus(options?: ApiRequestOptions): Promise {
+ return fetchJson(`${SYSTEM_BASE}/retention`, {
+ ...options,
+ cache: "no-store",
+ });
+}
+
+export function saveRetentionSettings(
+ settings: RetentionSettings,
+ options?: ApiRequestOptions,
+): Promise {
+ return fetchJson(`${SYSTEM_BASE}/retention`, {
+ ...options,
+ init: {
+ ...(options?.init ?? {}),
+ method: "PUT",
+ body: JSON.stringify(settings),
+ },
+ });
+}
diff --git a/apps/web/lib/settings-discovery/catalog/system.ts b/apps/web/lib/settings-discovery/catalog/system.ts
index f2f71da4bcc..82405c65736 100644
--- a/apps/web/lib/settings-discovery/catalog/system.ts
+++ b/apps/web/lib/settings-discovery/catalog/system.ts
@@ -9,6 +9,7 @@ export const SYSTEM_STORAGE_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/storage`;
export const SYSTEM_ABOUT_SETTINGS_HREF = `${SYSTEM_SETTINGS_HREF}/about`;
export const SYSTEM_SETTINGS_TARGETS = {
database: "setting-system-database",
+ retention: "setting-system-retention",
backups: "setting-system-backups",
logs: "setting-system-logs",
licenses: "setting-system-licenses",
@@ -59,6 +60,16 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [
targetId: SYSTEM_SETTINGS_TARGETS.database,
order: 621,
},
+ {
+ id: "system-retention",
+ kind: "section",
+ labelKey: "system:navRetention",
+ parentId: SYSTEM_DATA_STORAGE_DISCOVERY_ID,
+ groupId: "system",
+ href: SYSTEM_DATA_STORAGE_SETTINGS_HREF,
+ targetId: SYSTEM_SETTINGS_TARGETS.retention,
+ order: 622,
+ },
{
id: "system-backups",
kind: "section",
@@ -67,7 +78,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [
groupId: "system",
href: SYSTEM_DATA_STORAGE_SETTINGS_HREF,
targetId: SYSTEM_SETTINGS_TARGETS.backups,
- order: 622,
+ order: 623,
},
{
id: "system-logs",
@@ -77,7 +88,7 @@ export const SYSTEM_DISCOVERY_DEFINITIONS: SettingsDiscoveryDefinition[] = [
groupId: "system",
href: SYSTEM_DATA_STORAGE_SETTINGS_HREF,
targetId: SYSTEM_SETTINGS_TARGETS.logs,
- order: 623,
+ order: 624,
},
{
id: "system-storage",
diff --git a/apps/web/lib/state/slices/system/system-slice.test.ts b/apps/web/lib/state/slices/system/system-slice.test.ts
index ef24efea1e4..00e1103e73e 100644
--- a/apps/web/lib/state/slices/system/system-slice.test.ts
+++ b/apps/web/lib/state/slices/system/system-slice.test.ts
@@ -200,6 +200,30 @@ describe("system slice", () => {
expect(store.getState().system.database).toEqual(DB_STATS);
});
+ it("setSystemRetention stores the status", () => {
+ const store = makeStore();
+ expect(store.getState().system.retention).toBeNull();
+ const status = {
+ settings: {
+ enabled: true,
+ sweep_interval_hours: 6,
+ batch_limit: 5000,
+ routine_runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ runs: { window_days: 30, floor_per_owner: 50, warn_rows: 25000 },
+ run_events: { warn_rows: 250000 },
+ },
+ last_sweep: null,
+ skip_count: 0,
+ retained_counts: {
+ office_routine_runs: { state: "not_computed" as const, retained_count: 0, as_of: "" },
+ runs: { state: "not_computed" as const, retained_count: 0, as_of: "" },
+ run_events: { state: "not_computed" as const, retained_count: 0, as_of: "" },
+ },
+ };
+ store.getState().setSystemRetention(status);
+ expect(store.getState().system.retention).toEqual(status);
+ });
+
it("setSystemBackups marks the list as loaded", () => {
const store = makeStore();
store.getState().setSystemBackups([SNAPSHOT]);
diff --git a/apps/web/lib/state/slices/system/system-slice.ts b/apps/web/lib/state/slices/system/system-slice.ts
index 22ed5baeaa8..fc26162b758 100644
--- a/apps/web/lib/state/slices/system/system-slice.ts
+++ b/apps/web/lib/state/slices/system/system-slice.ts
@@ -6,6 +6,7 @@ export const defaultSystemState: SystemSliceState = {
info: null,
diskUsage: null,
database: null,
+ retention: null,
backups: { items: [], loaded: false },
updates: null,
jobs: {},
@@ -44,6 +45,10 @@ export const createSystemSlice: StateCreator<
set((draft) => {
draft.system.database = stats;
}),
+ setSystemRetention: (status) =>
+ set((draft) => {
+ draft.system.retention = status;
+ }),
setSystemBackups: (items) =>
set((draft) => {
draft.system.backups = { items, loaded: true };
diff --git a/apps/web/lib/state/slices/system/types.ts b/apps/web/lib/state/slices/system/types.ts
index 7135fdaf363..ecb21316e54 100644
--- a/apps/web/lib/state/slices/system/types.ts
+++ b/apps/web/lib/state/slices/system/types.ts
@@ -11,6 +11,7 @@ import type {
StorageOverviewResponse,
StoragePolicyResponse,
StorageQuarantineEntry,
+ RetentionStatus,
} from "@/lib/types/system";
export type SystemBackupsState = {
@@ -25,6 +26,7 @@ export type SystemSliceState = {
info: SystemInfo | null;
diskUsage: DiskUsageResponse | null;
database: DatabaseStats | null;
+ retention: RetentionStatus | null;
backups: SystemBackupsState;
updates: UpdatesResponse | null;
jobs: SystemJobsMap;
@@ -44,6 +46,7 @@ export type SystemSliceActions = {
setSystemInfo: (info: SystemInfo) => void;
setSystemDiskUsage: (usage: DiskUsageResponse) => void;
setSystemDatabase: (stats: DatabaseStats) => void;
+ setSystemRetention: (status: RetentionStatus) => void;
setSystemBackups: (items: SnapshotInfo[]) => void;
setSystemUpdates: (updates: UpdatesResponse) => void;
upsertSystemJob: (job: SystemJob) => void;
diff --git a/apps/web/lib/types/system.ts b/apps/web/lib/types/system.ts
index 09949372fbb..399bc00dd05 100644
--- a/apps/web/lib/types/system.ts
+++ b/apps/web/lib/types/system.ts
@@ -560,6 +560,78 @@ export interface StorageAdoptionResponse extends StorageSettingsResponse {
capabilities: StorageCapabilities;
}
+// --- Office run history retention ---------------------------------------
+
+export interface RetentionTableSettings {
+ window_days: number;
+ floor_per_owner: number;
+ warn_rows: number;
+}
+
+export interface RetentionRunEventsSettings {
+ warn_rows: number;
+}
+
+export interface RetentionSettings {
+ enabled: boolean;
+ sweep_interval_hours: number;
+ batch_limit: number;
+ routine_runs: RetentionTableSettings;
+ runs: RetentionTableSettings;
+ run_events: RetentionRunEventsSettings;
+}
+
+export interface RetentionTableSweepResult {
+ deleted: number;
+ backlog: boolean;
+ error: string;
+}
+
+export interface RetentionSweptTableResult extends RetentionTableSweepResult {
+ previewed: boolean;
+ would_delete: number;
+}
+
+export interface RetentionLastSweep {
+ started_at: string;
+ finished_at: string;
+ office_routine_runs: RetentionSweptTableResult;
+ runs: RetentionSweptTableResult;
+ run_events: RetentionTableSweepResult;
+ route_attempts: RetentionTableSweepResult;
+ run_skills: RetentionTableSweepResult;
+}
+
+export type RetentionCensusState = "not_computed" | "fresh" | "stale";
+
+export interface RetentionUnknownStatusCount {
+ status: string;
+ count: number;
+}
+
+export interface RetentionTableCensus {
+ state: RetentionCensusState;
+ retained_count: number;
+ as_of: string;
+ unknown_statuses?: RetentionUnknownStatusCount[];
+ top_routine_id?: string;
+ top_routine_share?: number;
+}
+
+export interface RetentionRetainedCounts {
+ office_routine_runs: RetentionTableCensus;
+ runs: RetentionTableCensus;
+ run_events: RetentionTableCensus;
+}
+
+export interface RetentionStatus {
+ settings: RetentionSettings;
+ last_sweep: RetentionLastSweep | null;
+ skip_count: number;
+ last_skip_at?: string;
+ retained_counts: RetentionRetainedCounts;
+}
+
export interface RestartCapability {
supported: boolean;
mode: "manual" | "supervisor" | string;
diff --git a/apps/web/src/locales/en/system.json b/apps/web/src/locales/en/system.json
index 63bda933697..b103768c767 100644
--- a/apps/web/src/locales/en/system.json
+++ b/apps/web/src/locales/en/system.json
@@ -207,8 +207,52 @@
"navFeatureToggles": "Feature Toggles",
"navLicenses": "Licenses",
"navLogs": "Logs",
+ "navRetention": "Run History Retention",
"navUpdates": "Updates",
"navUsers": "Users",
+ "retentionPageDescription": "Bounds how long Office run history (office_routine_runs, runs, and their satellite tables) stays in the database.",
+ "retentionAdminOnly": "Only an admin can change retention settings.",
+ "retentionLoadFailed": "Retention settings could not be loaded.",
+ "retentionSaveFailed": "Retention settings could not be saved.",
+ "retentionPolicyTitle": "Retention policy",
+ "retentionPolicyDescription": "A scheduled sweep, separate from the 5-second Office tick, deletes finished run history once it passes its retention window. Rows that represent current work in progress are never deleted, regardless of these settings.",
+ "retentionEnabledLabel": "Delete eligible run history",
+ "retentionEnabledDescription": "When off, Kandev never deletes office_routine_runs or runs rows. Retained-row counts below keep updating so you can see backlog build up before turning deletion back on.",
+ "retentionSweepIntervalLabel": "Sweep interval (hours)",
+ "retentionSweepIntervalHelp": "How often the sweep runs, from 1 to 168 hours (1 week).",
+ "retentionBatchLimitLabel": "Rows deleted per sweep",
+ "retentionBatchLimitHelp": "Maximum rows deleted per table in one sweep, from 100 to 100,000. Lower this if a sweep competes with other database load; a large backlog is cleared over several sweeps instead of one.",
+ "retentionRoutineRunsSectionTitle": "office_routine_runs",
+ "retentionRoutineRunsSectionDescription": "Routine execution records: one row per completed or failed routine run.",
+ "retentionRunsSectionTitle": "runs",
+ "retentionRunsSectionDescription": "Task run records: one row per completed, failed, or cancelled run.",
+ "retentionRunEventsSectionTitle": "run_events and satellite tables",
+ "retentionRunEventsSectionDescription": "run_events, office_run_route_attempts, and office_run_skills rows are deleted together with the run that owns them; they have no window or floor of their own.",
+ "retentionWindowDaysLabel": "Retention window (days)",
+ "retentionWindowDaysHelp": "Finished rows older than this many days become eligible for deletion, from 1 to 3650 days.",
+ "retentionFloorPerOwnerLabel": "Minimum kept per owner",
+ "retentionFloorPerOwnerHelp": "Always keeps at least this many of each owner's most recent finished rows, even past the retention window, from 0 to 10,000. Set to 0 to disable the floor.",
+ "retentionWarnRowsLabel": "Warn above this many retained rows",
+ "retentionWarnRowsHelp": "Raises a health warning once retained rows exceed this count. Set to 0 to disable the warning.",
+ "retentionRunEventsWarnRowsHelp": "Raises a health warning once total retained run_events rows exceed this count. Set to 0 to disable the warning.",
+ "retentionStatusTitle": "Retention status",
+ "retentionStatusDescription": "The most recent sweep outcome and the current retained-row counts for each thresholded table.",
+ "retentionNeverSweptMessage": "No sweep has run yet since the backend last started.",
+ "retentionSweepStartedAtLabel": "Started",
+ "retentionSweepFinishedAtLabel": "Finished",
+ "retentionDeletedLabel": "Deleted",
+ "retentionWouldDeleteLabel": "Would delete (preview)",
+ "retentionBacklogLabel": "Backlog: more eligible rows remain past this sweep's batch limit",
+ "retentionTableErrorLabel": "Error",
+ "retentionSkipCountLabel": "Sweeps skipped",
+ "retentionSkipCountHelp": "Counts sweeps that were due but did not run, most often because another Kandev process held the retention lock at the same time.",
+ "retentionLastSkipAtLabel": "Last skipped",
+ "retentionRetainedCountsTitle": "Retained rows",
+ "retentionCensusNotComputed": "Not yet measured",
+ "retentionCensusStale": "Stale: last measurement failed, showing the last successful count",
+ "retentionCensusAsOfLabel": "As of",
+ "retentionUnknownStatusesLabel": "Rows with an unrecognized status (not counted as history or live state)",
+ "retentionTopRoutineShareLabel": "Largest single routine's share of retained rows",
"backendReloadRequiredTitle": "Reload required",
"backendReloadRequiredBody": "Kandev restarted. Reload this page to continue. Reloading discards unsaved changes.",
"backendReloadRequiredAction": "Reload page",
diff --git a/apps/web/src/locales/pseudo/system.json b/apps/web/src/locales/pseudo/system.json
index 7711697a8e3..f90f569e914 100644
--- a/apps/web/src/locales/pseudo/system.json
+++ b/apps/web/src/locales/pseudo/system.json
@@ -207,8 +207,52 @@
"navFeatureToggles": "Ƒēàţũŕē Ţōĝĝĺēś",
"navLicenses": "Ĺĩćēńśēś",
"navLogs": "Ĺōĝś",
+ "navRetention": "Ŕũń Ĥĩśţōŕŷ Ŕēţēńţĩōń",
"navUpdates": "Ũƥďàţēś",
"navUsers": "Ũśēŕś",
+ "retentionPageDescription": "Ɓōũńďś ĥōŵ ĺōńĝ Ōƒƒĩćē ŕũń ĥĩśţōŕŷ (ōƒƒĩćē_ŕōũţĩńē_ŕũńś, ŕũńś, àńď ţĥēĩŕ śàţēĺĺĩţē ţàƀĺēś) śţàŷś ĩń ţĥē ďàţàƀàśē.",
+ "retentionAdminOnly": "Ōńĺŷ àń àďḿĩń ćàń ćĥàńĝē ŕēţēńţĩōń śēţţĩńĝś.",
+ "retentionLoadFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē ĺōàďēď.",
+ "retentionSaveFailed": "Ŕēţēńţĩōń śēţţĩńĝś ćōũĺď ńōţ ƀē śàvēď.",
+ "retentionPolicyTitle": "Ŕēţēńţĩōń ƥōĺĩćŷ",
+ "retentionPolicyDescription": "À śćĥēďũĺēď śŵēēƥ, śēƥàŕàţē ƒŕōḿ ţĥē 5-śēćōńď Ōƒƒĩćē ţĩćķ, ďēĺēţēś ƒĩńĩśĥēď ŕũń ĥĩśţōŕŷ ōńćē ĩţ ƥàśśēś ĩţś ŕēţēńţĩōń ŵĩńďōŵ. Ŕōŵś ţĥàţ ŕēƥŕēśēńţ ćũŕŕēńţ ŵōŕķ ĩń ƥŕōĝŕēśś àŕē ńēvēŕ ďēĺēţēď, ŕēĝàŕďĺēśś ōƒ ţĥēśē śēţţĩńĝś.",
+ "retentionEnabledLabel": "Ďēĺēţē ēĺĩĝĩƀĺē ŕũń ĥĩśţōŕŷ",
+ "retentionEnabledDescription": "Ŵĥēń ōƒƒ, Ķàńďēv ńēvēŕ ďēĺēţēś ōƒƒĩćē_ŕōũţĩńē_ŕũńś ōŕ ŕũńś ŕōŵś. Ŕēţàĩńēď-ŕōŵ ćōũńţś ƀēĺōŵ ķēēƥ ũƥďàţĩńĝ śō ŷōũ ćàń śēē ƀàćķĺōĝ ƀũĩĺď ũƥ ƀēƒōŕē ţũŕńĩńĝ ďēĺēţĩōń ƀàćķ ōń.",
+ "retentionSweepIntervalLabel": "Śŵēēƥ ĩńţēŕvàĺ (ĥōũŕś)",
+ "retentionSweepIntervalHelp": "Ĥōŵ ōƒţēń ţĥē śŵēēƥ ŕũńś, ƒŕōḿ 1 ţō 168 ĥōũŕś (1 ŵēēķ).",
+ "retentionBatchLimitLabel": "Ŕōŵś ďēĺēţēď ƥēŕ śŵēēƥ",
+ "retentionBatchLimitHelp": "Ḿàxĩḿũḿ ŕōŵś ďēĺēţēď ƥēŕ ţàƀĺē ĩń ōńē śŵēēƥ, ƒŕōḿ 100 ţō 100,000. Ĺōŵēŕ ţĥĩś ĩƒ à śŵēēƥ ćōḿƥēţēś ŵĩţĥ ōţĥēŕ ďàţàƀàśē ĺōàď; à ĺàŕĝē ƀàćķĺōĝ ĩś ćĺēàŕēď ōvēŕ śēvēŕàĺ śŵēēƥś ĩńśţēàď ōƒ ōńē.",
+ "retentionRoutineRunsSectionTitle": "ōƒƒĩćē_ŕōũţĩńē_ŕũńś",
+ "retentionRoutineRunsSectionDescription": "Ŕōũţĩńē ēxēćũţĩōń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď ōŕ ƒàĩĺēď ŕōũţĩńē ŕũń.",
+ "retentionRunsSectionTitle": "ŕũńś",
+ "retentionRunsSectionDescription": "Ţàśķ ŕũń ŕēćōŕďś: ōńē ŕōŵ ƥēŕ ćōḿƥĺēţēď, ƒàĩĺēď, ōŕ ćàńćēĺĺēď ŕũń.",
+ "retentionRunEventsSectionTitle": "ŕũń_ēvēńţś àńď śàţēĺĺĩţē ţàƀĺēś",
+ "retentionRunEventsSectionDescription": "ŕũń_ēvēńţś, ōƒƒĩćē_ŕũń_ŕōũţē_àţţēḿƥţś, àńď ōƒƒĩćē_ŕũń_śķĩĺĺś ŕōŵś àŕē ďēĺēţēď ţōĝēţĥēŕ ŵĩţĥ ţĥē ŕũń ţĥàţ ōŵńś ţĥēḿ; ţĥēŷ ĥàvē ńō ŵĩńďōŵ ōŕ ƒĺōōŕ ōƒ ţĥēĩŕ ōŵń.",
+ "retentionWindowDaysLabel": "Ŕēţēńţĩōń ŵĩńďōŵ (ďàŷś)",
+ "retentionWindowDaysHelp": "Ƒĩńĩśĥēď ŕōŵś ōĺďēŕ ţĥàń ţĥĩś ḿàńŷ ďàŷś ƀēćōḿē ēĺĩĝĩƀĺē ƒōŕ ďēĺēţĩōń, ƒŕōḿ 1 ţō 3650 ďàŷś.",
+ "retentionFloorPerOwnerLabel": "Ḿĩńĩḿũḿ ķēƥţ ƥēŕ ōŵńēŕ",
+ "retentionFloorPerOwnerHelp": "Àĺŵàŷś ķēēƥś àţ ĺēàśţ ţĥĩś ḿàńŷ ōƒ ēàćĥ ōŵńēŕ'ś ḿōśţ ŕēćēńţ ƒĩńĩśĥēď ŕōŵś, ēvēń ƥàśţ ţĥē ŕēţēńţĩōń ŵĩńďōŵ, ƒŕōḿ 0 ţō 10,000. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ƒĺōōŕ.",
+ "retentionWarnRowsLabel": "Ŵàŕń àƀōvē ţĥĩś ḿàńŷ ŕēţàĩńēď ŕōŵś",
+ "retentionWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ŕēţàĩńēď ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.",
+ "retentionRunEventsWarnRowsHelp": "Ŕàĩśēś à ĥēàĺţĥ ŵàŕńĩńĝ ōńćē ţōţàĺ ŕēţàĩńēď ŕũń_ēvēńţś ŕōŵś ēxćēēď ţĥĩś ćōũńţ. Śēţ ţō 0 ţō ďĩśàƀĺē ţĥē ŵàŕńĩńĝ.",
+ "retentionStatusTitle": "Ŕēţēńţĩōń śţàţũś",
+ "retentionStatusDescription": "Ţĥē ḿōśţ ŕēćēńţ śŵēēƥ ōũţćōḿē àńď ţĥē ćũŕŕēńţ ŕēţàĩńēď-ŕōŵ ćōũńţś ƒōŕ ēàćĥ ţĥŕēśĥōĺďēď ţàƀĺē.",
+ "retentionNeverSweptMessage": "Ńō śŵēēƥ ĥàś ŕũń ŷēţ śĩńćē ţĥē ƀàćķēńď ĺàśţ śţàŕţēď.",
+ "retentionSweepStartedAtLabel": "Śţàŕţēď",
+ "retentionSweepFinishedAtLabel": "Ƒĩńĩśĥēď",
+ "retentionDeletedLabel": "Ďēĺēţēď",
+ "retentionWouldDeleteLabel": "Ŵōũĺď ďēĺēţē (ƥŕēvĩēŵ)",
+ "retentionBacklogLabel": "Ɓàćķĺōĝ: ḿōŕē ēĺĩĝĩƀĺē ŕōŵś ŕēḿàĩń ƥàśţ ţĥĩś śŵēēƥ'ś ƀàţćĥ ĺĩḿĩţ",
+ "retentionTableErrorLabel": "Ēŕŕōŕ",
+ "retentionSkipCountLabel": "Śŵēēƥś śķĩƥƥēď",
+ "retentionSkipCountHelp": "Ćōũńţś śŵēēƥś ţĥàţ ŵēŕē ďũē ƀũţ ďĩď ńōţ ŕũń, ḿōśţ ōƒţēń ƀēćàũśē àńōţĥēŕ Ķàńďēv ƥŕōćēśś ĥēĺď ţĥē ŕēţēńţĩōń ĺōćķ àţ ţĥē śàḿē ţĩḿē.",
+ "retentionLastSkipAtLabel": "Ĺàśţ śķĩƥƥēď",
+ "retentionRetainedCountsTitle": "Ŕēţàĩńēď ŕōŵś",
+ "retentionCensusNotComputed": "Ńōţ ŷēţ ḿēàśũŕēď",
+ "retentionCensusStale": "Śţàĺē: ĺàśţ ḿēàśũŕēḿēńţ ƒàĩĺēď, śĥōŵĩńĝ ţĥē ĺàśţ śũććēśśƒũĺ ćōũńţ",
+ "retentionCensusAsOfLabel": "Àś ōƒ",
+ "retentionUnknownStatusesLabel": "Ŕōŵś ŵĩţĥ àń ũńŕēćōĝńĩźēď śţàţũś (ńōţ ćōũńţēď àś ĥĩśţōŕŷ ōŕ ĺĩvē śţàţē)",
+ "retentionTopRoutineShareLabel": "Ĺàŕĝēśţ śĩńĝĺē ŕōũţĩńē'ś śĥàŕē ōƒ ŕēţàĩńēď ŕōŵś",
"backendReloadRequiredTitle": "Ŕēĺōàď ŕēqũĩŕēď",
"backendReloadRequiredBody": "Ķàńďēv ŕēśţàŕţēď. Ŕēĺōàď ţĥĩś ƥàĝē ţō ćōńţĩńũē. Ŕēĺōàďĩńĝ ďĩśćàŕďś ũńśàvēď ćĥàńĝēś.",
"backendReloadRequiredAction": "Ŕēĺōàď ƥàĝē",
diff --git a/apps/web/src/locales/pt-pt/system.json b/apps/web/src/locales/pt-pt/system.json
index dc9122adfaf..097df6de1ca 100644
--- a/apps/web/src/locales/pt-pt/system.json
+++ b/apps/web/src/locales/pt-pt/system.json
@@ -202,8 +202,52 @@
"navFeatureToggles": "Interruptores de funcionalidades",
"navLicenses": "Licenças",
"navLogs": "Registos",
+ "navRetention": "Retenção do histórico de execuções",
"navUpdates": "Atualizações",
"navUsers": "Utilizadores",
+ "retentionPageDescription": "Limita durante quanto tempo o histórico de execuções do Office (office_routine_runs, runs e as respetivas tabelas satélite) permanece na base de dados.",
+ "retentionAdminOnly": "Só um administrador pode alterar as definições de retenção.",
+ "retentionLoadFailed": "Não foi possível carregar as definições de retenção.",
+ "retentionSaveFailed": "Não foi possível guardar as definições de retenção.",
+ "retentionPolicyTitle": "Política de retenção",
+ "retentionPolicyDescription": "Uma limpeza agendada, separada do ciclo do Office de 5 segundos, elimina o histórico de execuções concluído assim que ultrapassa a sua janela de retenção. As linhas que representam trabalho em curso nunca são eliminadas, independentemente destas definições.",
+ "retentionEnabledLabel": "Eliminar histórico de execuções elegível",
+ "retentionEnabledDescription": "Quando desativado, o Kandev nunca elimina linhas de office_routine_runs ou runs. As contagens de linhas retidas abaixo continuam a ser atualizadas para poder ver a acumulação de atrasos antes de reativar a eliminação.",
+ "retentionSweepIntervalLabel": "Intervalo de limpeza (horas)",
+ "retentionSweepIntervalHelp": "Frequência com que a limpeza é executada, entre 1 e 168 horas (1 semana).",
+ "retentionBatchLimitLabel": "Linhas eliminadas por limpeza",
+ "retentionBatchLimitHelp": "Número máximo de linhas eliminadas por tabela numa limpeza, entre 100 e 100 000. Reduza este valor se a limpeza competir com outra carga na base de dados; um atraso grande é eliminado ao longo de várias limpezas em vez de uma só.",
+ "retentionRoutineRunsSectionTitle": "office_routine_runs",
+ "retentionRoutineRunsSectionDescription": "Registos de execução de rotinas: uma linha por cada execução de rotina concluída ou falhada.",
+ "retentionRunsSectionTitle": "runs",
+ "retentionRunsSectionDescription": "Registos de execução de tarefas: uma linha por cada execução concluída, falhada ou cancelada.",
+ "retentionRunEventsSectionTitle": "run_events e tabelas satélite",
+ "retentionRunEventsSectionDescription": "As linhas de run_events, office_run_route_attempts e office_run_skills são eliminadas em conjunto com a execução que as possui; não têm janela nem limite mínimo próprios.",
+ "retentionWindowDaysLabel": "Janela de retenção (dias)",
+ "retentionWindowDaysHelp": "As linhas concluídas com mais deste número de dias tornam-se elegíveis para eliminação, entre 1 e 3650 dias.",
+ "retentionFloorPerOwnerLabel": "Mínimo mantido por proprietário",
+ "retentionFloorPerOwnerHelp": "Mantém sempre pelo menos este número de linhas concluídas mais recentes de cada proprietário, mesmo além da janela de retenção, entre 0 e 10 000. Defina 0 para desativar este limite mínimo.",
+ "retentionWarnRowsLabel": "Avisar acima deste número de linhas retidas",
+ "retentionWarnRowsHelp": "Gera um aviso de saúde quando as linhas retidas ultrapassam este número. Defina 0 para desativar o aviso.",
+ "retentionRunEventsWarnRowsHelp": "Gera um aviso de saúde quando o total de linhas de run_events retidas ultrapassa este número. Defina 0 para desativar o aviso.",
+ "retentionStatusTitle": "Estado da retenção",
+ "retentionStatusDescription": "O resultado da limpeza mais recente e as contagens atuais de linhas retidas para cada tabela com limiar definido.",
+ "retentionNeverSweptMessage": "Ainda não foi executada nenhuma limpeza desde o último arranque do backend.",
+ "retentionSweepStartedAtLabel": "Iniciada",
+ "retentionSweepFinishedAtLabel": "Concluída",
+ "retentionDeletedLabel": "Eliminadas",
+ "retentionWouldDeleteLabel": "Seriam eliminadas (pré-visualização)",
+ "retentionBacklogLabel": "Atraso: ainda há mais linhas elegíveis além do limite desta limpeza",
+ "retentionTableErrorLabel": "Erro",
+ "retentionSkipCountLabel": "Limpezas ignoradas",
+ "retentionSkipCountHelp": "Conta as limpezas que deviam ter sido executadas mas não o foram, geralmente porque outro processo do Kandev detinha o bloqueio de retenção ao mesmo tempo.",
+ "retentionLastSkipAtLabel": "Última ignorada em",
+ "retentionRetainedCountsTitle": "Linhas retidas",
+ "retentionCensusNotComputed": "Ainda não medido",
+ "retentionCensusStale": "Desatualizado: a última medição falhou, a mostrar a última contagem bem-sucedida",
+ "retentionCensusAsOfLabel": "Referente a",
+ "retentionUnknownStatusesLabel": "Linhas com um estado não reconhecido (não contadas como histórico nem como estado ativo)",
+ "retentionTopRoutineShareLabel": "Proporção de linhas retidas pertencentes à maior rotina individual",
"backendReloadRequiredTitle": "É necessário recarregar",
"backendReloadRequiredBody": "O Kandev foi reiniciado. Recarregue esta página para continuar. O recarregamento elimina as alterações não guardadas.",
"backendReloadRequiredAction": "Recarregar página",
diff --git a/apps/web/src/locales/zh-cn/system.json b/apps/web/src/locales/zh-cn/system.json
index 333cbe57e8c..1361e839d65 100644
--- a/apps/web/src/locales/zh-cn/system.json
+++ b/apps/web/src/locales/zh-cn/system.json
@@ -204,8 +204,52 @@
"navFeatureToggles": "功能开关",
"navLicenses": "许可证",
"navLogs": "日志",
+ "navRetention": "运行历史保留",
"navUpdates": "更新",
"navUsers": "用户",
+ "retentionPageDescription": "限制 Office 运行历史(office_routine_runs、runs 及其附属表)在数据库中保留的时长。",
+ "retentionAdminOnly": "只有管理员才能更改保留设置。",
+ "retentionLoadFailed": "无法加载保留设置。",
+ "retentionSaveFailed": "无法保存保留设置。",
+ "retentionPolicyTitle": "保留策略",
+ "retentionPolicyDescription": "一个与 5 秒 Office 心跳独立的定时清理任务,会在已完成的运行历史超出其保留窗口后将其删除。代表当前进行中工作的行永远不会被删除,与这些设置无关。",
+ "retentionEnabledLabel": "删除符合条件的运行历史",
+ "retentionEnabledDescription": "关闭时,Kandev 永远不会删除 office_routine_runs 或 runs 中的行。下方的保留行数仍会持续更新,方便你在重新启用删除前查看积压情况。",
+ "retentionSweepIntervalLabel": "清理间隔(小时)",
+ "retentionSweepIntervalHelp": "清理任务运行的频率,介于 1 到 168 小时(1 周)之间。",
+ "retentionBatchLimitLabel": "每次清理删除的行数",
+ "retentionBatchLimitHelp": "每次清理中每张表最多删除的行数,介于 100 到 100000 之间。如果清理与其他数据库负载相互争用,可以调低此值;较大的积压会分多次清理逐步处理,而不是一次完成。",
+ "retentionRoutineRunsSectionTitle": "office_routine_runs",
+ "retentionRoutineRunsSectionDescription": "例行任务执行记录:每完成或失败一次例行任务运行对应一行。",
+ "retentionRunsSectionTitle": "runs",
+ "retentionRunsSectionDescription": "任务运行记录:每完成、失败或取消一次运行对应一行。",
+ "retentionRunEventsSectionTitle": "run_events 及附属表",
+ "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行会随其所属的运行一起删除;它们没有自己的窗口或下限设置。",
+ "retentionWindowDaysLabel": "保留窗口(天)",
+ "retentionWindowDaysHelp": "已完成且超过此天数的行将符合删除条件,介于 1 到 3650 天之间。",
+ "retentionFloorPerOwnerLabel": "每个所有者的最少保留数",
+ "retentionFloorPerOwnerHelp": "始终至少保留每个所有者最近完成的这么多行,即使超出保留窗口,介于 0 到 10000 之间。设为 0 可停用此下限。",
+ "retentionWarnRowsLabel": "超过此保留行数时发出警告",
+ "retentionWarnRowsHelp": "当保留行数超过此数值时触发健康警告。设为 0 可停用该警告。",
+ "retentionRunEventsWarnRowsHelp": "当保留的 run_events 总行数超过此数值时触发健康警告。设为 0 可停用该警告。",
+ "retentionStatusTitle": "保留状态",
+ "retentionStatusDescription": "最近一次清理的结果,以及每张设有阈值的表当前的保留行数。",
+ "retentionNeverSweptMessage": "自后端上次启动以来,尚未运行过清理任务。",
+ "retentionSweepStartedAtLabel": "开始于",
+ "retentionSweepFinishedAtLabel": "结束于",
+ "retentionDeletedLabel": "已删除",
+ "retentionWouldDeleteLabel": "预计删除(预览)",
+ "retentionBacklogLabel": "积压:超出本次清理批量上限的符合条件的行仍然存在",
+ "retentionTableErrorLabel": "错误",
+ "retentionSkipCountLabel": "已跳过的清理次数",
+ "retentionSkipCountHelp": "统计原本应该运行但未运行的清理次数,通常是因为同一时间有另一个 Kandev 进程持有保留锁。",
+ "retentionLastSkipAtLabel": "上次跳过于",
+ "retentionRetainedCountsTitle": "保留的行数",
+ "retentionCensusNotComputed": "尚未统计",
+ "retentionCensusStale": "已过期:上次统计失败,显示的是上一次成功的计数",
+ "retentionCensusAsOfLabel": "统计时间",
+ "retentionUnknownStatusesLabel": "状态无法识别的行(既不计入历史,也不计入当前状态)",
+ "retentionTopRoutineShareLabel": "占保留行数比例最高的单个例行任务",
"backendReloadRequiredTitle": "需要重新加载",
"backendReloadRequiredBody": "Kandev 已重启。请重新加载此页面以继续。重新加载会丢弃未保存的更改。",
"backendReloadRequiredAction": "重新加载页面",
diff --git a/apps/web/src/locales/zh-hk/system.json b/apps/web/src/locales/zh-hk/system.json
index 38563613bbe..3f65e70e084 100644
--- a/apps/web/src/locales/zh-hk/system.json
+++ b/apps/web/src/locales/zh-hk/system.json
@@ -204,8 +204,52 @@
"navFeatureToggles": "功能開關",
"navLicenses": "許可證",
"navLogs": "日誌",
+ "navRetention": "執行歷史保留",
"navUpdates": "更新",
"navUsers": "用戶",
+ "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在數據庫中保留的時長。",
+ "retentionAdminOnly": "只有管理員才能更改保留設定。",
+ "retentionLoadFailed": "無法載入保留設定。",
+ "retentionSaveFailed": "無法儲存保留設定。",
+ "retentionPolicyTitle": "保留策略",
+ "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留窗口後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。",
+ "retentionEnabledLabel": "刪除符合條件的執行歷史",
+ "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啓用刪除前查看積壓情況。",
+ "retentionSweepIntervalLabel": "清理間隔(小時)",
+ "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。",
+ "retentionBatchLimitLabel": "每次清理刪除的行數",
+ "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他數據庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。",
+ "retentionRoutineRunsSectionTitle": "office_routine_runs",
+ "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。",
+ "retentionRunsSectionTitle": "runs",
+ "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。",
+ "retentionRunEventsSectionTitle": "run_events 及附屬表",
+ "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的窗口或下限設定。",
+ "retentionWindowDaysLabel": "保留窗口(天)",
+ "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。",
+ "retentionFloorPerOwnerLabel": "每個所有者的最少保留數",
+ "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留窗口,介於 0 到 10000 之間。設為 0 可停用此下限。",
+ "retentionWarnRowsLabel": "超過此保留行數時發出警告",
+ "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。",
+ "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。",
+ "retentionStatusTitle": "保留狀態",
+ "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。",
+ "retentionNeverSweptMessage": "自後端上次啓動以來,尚未執行過清理任務。",
+ "retentionSweepStartedAtLabel": "開始於",
+ "retentionSweepFinishedAtLabel": "結束於",
+ "retentionDeletedLabel": "已刪除",
+ "retentionWouldDeleteLabel": "預計刪除(預覽)",
+ "retentionBacklogLabel": "積壓:超出本次清理批量上限的符合條件的行仍然存在",
+ "retentionTableErrorLabel": "錯誤",
+ "retentionSkipCountLabel": "已跳過的清理次數",
+ "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 程序持有保留鎖。",
+ "retentionLastSkipAtLabel": "上次跳過於",
+ "retentionRetainedCountsTitle": "保留的行數",
+ "retentionCensusNotComputed": "尚未統計",
+ "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數",
+ "retentionCensusAsOfLabel": "統計時間",
+ "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)",
+ "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務",
"backendReloadRequiredTitle": "需要重新載入",
"backendReloadRequiredBody": "Kandev 已重啓。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。",
"backendReloadRequiredAction": "重新載入頁面",
diff --git a/apps/web/src/locales/zh-tw/system.json b/apps/web/src/locales/zh-tw/system.json
index 8aec9747198..192eec0cddb 100644
--- a/apps/web/src/locales/zh-tw/system.json
+++ b/apps/web/src/locales/zh-tw/system.json
@@ -204,8 +204,52 @@
"navFeatureToggles": "功能開關",
"navLicenses": "許可證",
"navLogs": "日誌",
+ "navRetention": "執行歷史保留",
"navUpdates": "更新",
"navUsers": "使用者",
+ "retentionPageDescription": "限制 Office 執行歷史(office_routine_runs、runs 及其附屬表)在資料庫中保留的時長。",
+ "retentionAdminOnly": "只有管理員才能更改保留設定。",
+ "retentionLoadFailed": "無法載入保留設定。",
+ "retentionSaveFailed": "無法儲存保留設定。",
+ "retentionPolicyTitle": "保留策略",
+ "retentionPolicyDescription": "一個與 5 秒 Office 心跳獨立的定時清理任務,會在已完成的執行歷史超出其保留視窗後將其刪除。代表目前進行中工作的行永遠不會被刪除,與這些設定無關。",
+ "retentionEnabledLabel": "刪除符合條件的執行歷史",
+ "retentionEnabledDescription": "關閉時,Kandev 永遠不會刪除 office_routine_runs 或 runs 中的行。下方的保留行數仍會持續更新,方便你在重新啟用刪除前檢視積壓情況。",
+ "retentionSweepIntervalLabel": "清理間隔(小時)",
+ "retentionSweepIntervalHelp": "清理任務執行的頻率,介於 1 到 168 小時(1 周)之間。",
+ "retentionBatchLimitLabel": "每次清理刪除的行數",
+ "retentionBatchLimitHelp": "每次清理中每張表最多刪除的行數,介於 100 到 100000 之間。如果清理與其他資料庫負載相互爭用,可以調低此值;較大的積壓會分多次清理逐步處理,而不是一次完成。",
+ "retentionRoutineRunsSectionTitle": "office_routine_runs",
+ "retentionRoutineRunsSectionDescription": "例行任務執行記錄:每完成或失敗一次例行任務執行對應一行。",
+ "retentionRunsSectionTitle": "runs",
+ "retentionRunsSectionDescription": "任務執行記錄:每完成、失敗或取消一次執行對應一行。",
+ "retentionRunEventsSectionTitle": "run_events 及附屬表",
+ "retentionRunEventsSectionDescription": "run_events、office_run_route_attempts 和 office_run_skills 中的行會隨其所屬的執行一起刪除;它們沒有自己的視窗或下限設定。",
+ "retentionWindowDaysLabel": "保留視窗(天)",
+ "retentionWindowDaysHelp": "已完成且超過此天數的行將符合刪除條件,介於 1 到 3650 天之間。",
+ "retentionFloorPerOwnerLabel": "每個所有者的最少保留數",
+ "retentionFloorPerOwnerHelp": "始終至少保留每個所有者最近完成的這麼多行,即使超出保留視窗,介於 0 到 10000 之間。設為 0 可停用此下限。",
+ "retentionWarnRowsLabel": "超過此保留行數時發出警告",
+ "retentionWarnRowsHelp": "當保留行數超過此數值時觸發健康警告。設為 0 可停用該警告。",
+ "retentionRunEventsWarnRowsHelp": "當保留的 run_events 總行數超過此數值時觸發健康警告。設為 0 可停用該警告。",
+ "retentionStatusTitle": "保留狀態",
+ "retentionStatusDescription": "最近一次清理的結果,以及每張設有閾值的表目前的保留行數。",
+ "retentionNeverSweptMessage": "自後端上次啟動以來,尚未執行過清理任務。",
+ "retentionSweepStartedAtLabel": "開始於",
+ "retentionSweepFinishedAtLabel": "結束於",
+ "retentionDeletedLabel": "已刪除",
+ "retentionWouldDeleteLabel": "預計刪除(預覽)",
+ "retentionBacklogLabel": "積壓:超出本次清理批次上限的符合條件的行仍然存在",
+ "retentionTableErrorLabel": "錯誤",
+ "retentionSkipCountLabel": "已跳過的清理次數",
+ "retentionSkipCountHelp": "統計原本應該執行但未執行的清理次數,通常是因為同一時間有另一個 Kandev 處理程序持有保留鎖。",
+ "retentionLastSkipAtLabel": "上次跳過於",
+ "retentionRetainedCountsTitle": "保留的行數",
+ "retentionCensusNotComputed": "尚未統計",
+ "retentionCensusStale": "已過期:上次統計失敗,顯示的是上一次成功的計數",
+ "retentionCensusAsOfLabel": "統計時間",
+ "retentionUnknownStatusesLabel": "狀態無法識別的行(既不計入歷史,也不計入目前狀態)",
+ "retentionTopRoutineShareLabel": "佔保留行數比例最高的單個例行任務",
"backendReloadRequiredTitle": "需要重新載入",
"backendReloadRequiredBody": "Kandev 已重啟。請重新載入此頁面以繼續。重新載入會丟棄未儲存的更改。",
"backendReloadRequiredAction": "重新載入頁面",
diff --git a/docs/plans/run-history-retention/plan.md b/docs/plans/run-history-retention/plan.md
new file mode 100644
index 00000000000..36b58620542
--- /dev/null
+++ b/docs/plans/run-history-retention/plan.md
@@ -0,0 +1,64 @@
+---
+created: 2026-09-09
+status: complete
+requirements:
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-001
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-002
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-003
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-004
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-005
+system_design:
+ - ../../specs/office/system-design/run-history-retention.md
+ - ../../specs/office/system-design/run-history-retention-operations.md
+legacy_specs: []
+---
+
+# Implementation Plan: Office Run History Retention
+
+## Overview
+
+Office writes `office_routine_runs` and `run_events` rows on every routine firing and
+run lifecycle transition, and nothing ever deletes them on a schedule. A `*/5 * * * *`
+routine produces roughly 105,000 `office_routine_runs` rows a year; one reference
+install had already accumulated 323 consecutive `coalesced` rows from a routine that
+was doing nothing useful. This plan adds one scheduled sweep, on its own interval
+(never the 5s Office tick), that bounds `office_routine_runs`, `runs`, and their
+satellites (`run_events`, `office_run_route_attempts`, `office_run_skills`) by age,
+with a per-owner floor, identical behavior on SQLite and PostgreSQL, and an operator
+surface (Settings > System > Data & Logs) that reports policy, counts, previews, and
+warnings before and while rows are deleted.
+
+The two halves of the contract are split the same way the specs are split: the sweep
+itself, its eligibility rules, and engine parity are
+[run history retention](../../specs/office/system-design/run-history-retention.md)
+(REQ-001, REQ-002, REQ-005); the settings record, preview marker, health warnings, and
+System page surface are
+[run history retention operations](../../specs/office/system-design/run-history-retention-operations.md)
+(REQ-003, REQ-004).
+
+## Scope
+
+### In scope
+
+- A `internal/office/retention` package: settings store, eligibility/count/delete
+ queries shared by preview and delete, a session-scoped PostgreSQL advisory lock with
+ a SQLite single-process equivalent, a scheduler goroutine on its own interval, and an
+ HTTP handler for `GET`/`PUT /api/v1/system/retention`.
+- Status-only classification of history vs. live state for both `office_routine_runs`
+ and `runs`, a per-owner floor, oldest-first chunked batch deletion, and satellite rows
+ deleted in the same transaction as their parent run.
+- A per-table preview (report, delete nothing) on each table's first evaluation, and
+ `health.Issue` warnings before the cap and on sweep/count failure.
+- A `RetentionSettingsCard` on Settings > System > Data & Logs showing policy, retained
+ counts, preview state, last sweep, and backlog/error warnings.
+- Expression indexes serving the sweep's filter/order on both engines.
+
+### Out of scope
+
+- Filesystem/container cleanup (owned by storage maintenance).
+- Routine or workspace deletion (already deletes runs structurally; unaffected).
+- Any change to run lifecycle, routine dispatch, or task/session/checkout data.
+
+## Tasks
+
+- [x] [Task 01: Bound run history with a scheduled retention sweep](task-01-bound-run-history-with-retention-sweep.md)
diff --git a/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md
new file mode 100644
index 00000000000..7d7c2c15465
--- /dev/null
+++ b/docs/plans/run-history-retention/task-01-bound-run-history-with-retention-sweep.md
@@ -0,0 +1,183 @@
+---
+id: "01-bound-run-history-with-retention-sweep"
+title: "Bound Office run history with a scheduled retention sweep"
+status: done
+wave: 1
+depends_on: []
+plan: "plan.md"
+requirements:
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-001
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-002
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-003
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-004
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-005
+acceptance_criteria:
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.1
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.2
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.3
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.4
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.5
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.6
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.7
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.8
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.9
+ - AC-OFFICE-RUN-HISTORY-RETENTION-001.10
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.1
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.2
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.3
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.4
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.5
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.6
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.7
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.8
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.9
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.10
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.11
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.12
+ - AC-OFFICE-RUN-HISTORY-RETENTION-002.13
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.1
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.2
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.3
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.4
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.5
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.6
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.7
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.8
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.9
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.10
+ - AC-OFFICE-RUN-HISTORY-RETENTION-003.11
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.1
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.2
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.3
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.4
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.5
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.6
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.7
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.8
+ - AC-OFFICE-RUN-HISTORY-RETENTION-004.9
+ - AC-OFFICE-RUN-HISTORY-RETENTION-005.1
+ - AC-OFFICE-RUN-HISTORY-RETENTION-005.2
+ - AC-OFFICE-RUN-HISTORY-RETENTION-005.3
+ - AC-OFFICE-RUN-HISTORY-RETENTION-005.4
+ - AC-OFFICE-RUN-HISTORY-RETENTION-005.5
+system_design:
+ - ../../specs/office/system-design/run-history-retention.md
+ - ../../specs/office/system-design/run-history-retention-operations.md
+---
+
+# Task 01: Bound Office Run History with a Scheduled Retention Sweep
+
+## Summary
+
+Add `internal/office/retention`: a settings-backed sweep that ages out
+`office_routine_runs` and `runs` history (plus their satellites) on its own
+interval, a per-owner floor, status-only live/history classification, a preview
+pass before first deletion, engine-identical behavior on SQLite and PostgreSQL,
+and a `GET`/`PUT /api/v1/system/retention` operator surface with a matching
+Settings > System > Data & Logs card.
+
+## In scope
+
+- Eligibility, count, and delete queries shared by preview and the real sweep,
+ keyed on `COALESCE(completed_at, created_at)` / `COALESCE(finished_at,
+ requested_at)`, oldest-first, chunked across statements.
+- A session-scoped PostgreSQL advisory lock (with a SQLite single-process
+ equivalent) so only one backend sweeps at a time.
+- A per-owner floor (newest N rows retained regardless of age) applied to
+ count, preview, and delete identically, re-asserted in the DELETE statement.
+- Satellite deletion (`run_events`, `office_run_route_attempts`,
+ `office_run_skills`) in the same transaction as their parent run.
+- `health.Issue` warnings for approaching the backlog cap and for count/sweep
+ failure; a one-time preview per table before any row is deleted.
+- `RetentionSettingsCard` on Settings > System > Data & Logs: policy, retained
+ counts, preview/backlog state, last sweep, errors.
+- Expression indexes serving the sweep's filter/order on both engines.
+
+## Out of scope
+
+- Filesystem/container artifact cleanup (storage maintenance owns that).
+- Routine/workspace deletion cascades (already delete runs structurally).
+- Any run lifecycle, routing, or task/session/checkout behavior change.
+
+## Acceptance
+
+- History rows (terminal `office_routine_runs`/`runs` statuses) older than the
+ configured window are deleted in batches; live-state rows (`received`,
+ `task_created`, `queued`, `claimed`) are never age-pruned; an unrecognized
+ status fails safe as live state and warns.
+- The newest `floor` rows per owner survive regardless of age.
+- Each table's first sweep is a preview: it reports would-delete counts and
+ deletes nothing; a later sweep performs real deletion.
+- Behavior, including the advisory lock and batch chunking, is identical on
+ SQLite and PostgreSQL.
+- `GET`/`PUT /api/v1/system/retention` read/write policy and report current
+ counts, preview state, last sweep outcome, and backlog/error warnings.
+
+## Verification
+
+```bash
+cd apps/backend && go test -race -count=1 ./internal/office/retention/... ./internal/office/repository/sqlite/...
+cd apps/backend && KANDEV_TEST_POSTGRES_DSN= go test -race -count=1 -v ./internal/office/retention/...
+cd apps/backend && golangci-lint run ./internal/office/retention/...
+cd apps/web && pnpm run typecheck
+cd apps && pnpm --filter @kandev/web test -- --run retention-settings-card system-route-copy
+cd apps/web && pnpm e2e:run -- --project chromium --grep "System retention settings"
+```
+
+## Files likely touched
+
+- `apps/backend/internal/office/retention/*.go`
+- `apps/backend/internal/office/repository/sqlite/base_migrations.go`,
+ `retention_indexes_test.go`, `retention_indexes_postgres_test.go`
+- `apps/backend/internal/backendapp/*` (scheduler/handler wiring)
+- `apps/web/src/**/retention-settings-card.tsx` and its tests
+- `docs/specs/office/requirements/run-history-retention*.md`,
+ `docs/specs/office/system-design/run-history-retention*.md`
+- `docs/public/operations.md`
+
+## Dependencies
+
+None.
+
+## Risks
+
+- A dialect-sensitive query bug that only reproduces on PostgreSQL (mitigated
+ by an environment-gated Postgres suite covering the advisory lock, the
+ EvalPlanQual TOCTOU window, and batch chunking).
+- A too-aggressive window or floor deleting rows an operator still needed
+ (mitigated by the default 30-day window, the per-owner floor, and the
+ preview-before-delete behavior).
+
+## Parallelism
+
+`sequential`
+
+## Inputs
+
+- `REQ-OFFICE-RUN-HISTORY-RETENTION-001` through `-005`.
+- [run history retention](../../specs/office/system-design/run-history-retention.md)
+ and
+ [run history retention operations](../../specs/office/system-design/run-history-retention-operations.md).
+- The reference install's 323 consecutive `coalesced` routine-run rows (28
+ days, one bricked routine) cited in the requirements' Overview.
+
+## Results
+
+- Implemented `internal/office/retention` (settings store, `Sweeper`,
+ `Scheduler`, `CensusTracker`, PostgreSQL advisory lock with SQLite
+ equivalent, `Handler` for `GET`/`PUT /api/v1/system/retention`) plus the
+ `RetentionSettingsCard` on Settings > System > Data & Logs.
+- Full backend gauntlet green: `go build ./...`, `go vet`, `gofmt -l`,
+ `go test -race ./internal/office/retention/...` (SQLite), the
+ PostgreSQL-gated suite against a real scratch instance (125 tests, 0
+ skipped, all PASS, covering the advisory lock, the EvalPlanQual
+ concurrent-resurrection TOCTOU regression, and chunked batch deletes),
+ `golangci-lint run ./...` (0 issues).
+- Frontend: `pnpm run typecheck`, the retention card and System route-copy
+ Vitest suites, and a scoped Playwright run (`--grep "System retention
+ settings"`, 2 passed).
+- Delivered as PR [#3566](https://github.com/kdlbs/kandev/pull/3566)
+ ("feat(office): bound run history growth with a scheduled retention
+ sweep"), through four Build rounds and four Review rounds; remaining
+ non-blocking test-rigor gaps and CI-wiring follow-ups are tracked on the
+ linked follow-up card rather than blocking this PR.
diff --git a/docs/public/operations.md b/docs/public/operations.md
index 7ce3be85abd..58d19172a06 100644
--- a/docs/public/operations.md
+++ b/docs/public/operations.md
@@ -348,6 +348,34 @@ the Kandev service first provides the clearest maintenance boundary.
+## Office run history retention
+
+Open **Settings > System > Data & Logs** to manage automatic deletion of old
+Office run history. Deletion is enabled by default. The first sweep starts five
+minutes after the backend starts or after you enable deletion.
+
+The first sweep for each history table is a preview. It reports the rows that
+would be deleted and removes no rows. A later sweep can delete eligible rows.
+Deletion is permanent. Back up the database before you enable deletion if you
+need to keep old history outside the configured window.
+
+The retention window controls the age of rows that can be deleted. The minimum
+kept per owner control keeps the newest rows for each routine or agent, even
+when those rows are older than the window. A floor of zero removes this extra
+protection. Run event, route attempt, and skill rows are deleted with their
+parent run.
+
+The page shows the current policy, retained row counts, preview results, the
+last sweep, and any backlog or errors. Counts continue to update when deletion
+is disabled, so you can monitor growth before you enable it again.
+
+To disable automatic deletion:
+
+1. Open **Settings > System > Data & Logs**.
+2. Clear **Delete eligible run history**.
+3. Select **Save changes**.
+4. Check the retention status. It must show that deletion is disabled.
+
## Database operation
> **Single-owner rule:** SQLite uses one writer connection in WAL mode; only one Kandev backend should own the file. **Factory reset** is destructive and removes managed data after creating a pre-reset backup.
diff --git a/docs/specs/office/requirements/run-history-retention-operations.md b/docs/specs/office/requirements/run-history-retention-operations.md
new file mode 100644
index 00000000000..394f70290e9
--- /dev/null
+++ b/docs/specs/office/requirements/run-history-retention-operations.md
@@ -0,0 +1,214 @@
+---
+status: draft
+system: office
+created: 2026-09-09
+owners:
+ - kandev
+---
+
+# Office Run History Retention Operations Requirements
+
+## Overview
+
+[Run history retention](run-history-retention.md) defines which Office run
+history rows may be deleted and how the sweep that deletes them behaves. This
+document defines the other half of that contract: what an operator is told
+before the first row is removed, what they are told while the tables grow, what
+they can configure, and what they can read about the last sweep.
+
+The split follows the ownership boundary the retention contract already draws.
+Office owns the rows and the deletion policy because they are Office primitives.
+The System pages own the operator surface those decisions are reported through,
+and that surface is a separate contract with a separate consumer: an operator
+reading a settings page and a health card, rather than a scheduler deleting
+rows. Splitting here keeps each document inside its size limit without cutting
+either contract.
+
+The requirement IDs continue the retention capability's sequence rather than
+starting a new one, because these are the same capability's requirements viewed
+from the operator's side.
+
+## Terminology
+
+Terms are defined once, in
+[run history retention](run-history-retention.md#terminology), and used here
+with the same meaning. The ones this document leans on most:
+
+- **Retention sweep**, **preview**, and **retained count**.
+- **Swept tables** (`office_routine_runs`, `runs`), **reported tables** (those
+ two plus the three run satellites), and **thresholded tables**
+ (`office_routine_runs`, `runs`, `run_events`). "Per table" always names one of
+ these three sets, never an unqualified "table".
+
+## Requirements
+
+### REQ-OFFICE-RUN-HISTORY-RETENTION-003: Warn before deleting, and warn while growing
+
+**Intent:** An operator learns what retention is about to remove before it
+removes anything, and learns that history is growing past a threshold while
+there is still time to widen the window or fix the routine.
+
+As an operator upgrading an install with a year of run history, I want to be
+told what the first sweep would delete before it deletes it, so that I can widen
+the retention window first if that history matters to me.
+
+#### Acceptance criteria
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.1:** The first evaluation of each swept table
+ on a database shall be a preview for that table: it evaluates the policy,
+ reports the number of rows it would delete from that table, and deletes
+ nothing from it. A table that has not completed a preview never deletes,
+ regardless of what any other table has done.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.2:** When a preview sweep would delete
+ at least one row, the system shall emit an operator-visible warning naming
+ each table, its would-delete count, and the configured retention window, and
+ stating that deletion begins at the next scheduled sweep.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.3:** When a preview sweep would delete
+ no rows, the system shall not emit a warning.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.4:** The system shall run a preview at most
+ once per swept table per database. A table's preview shall be recorded only
+ when that table's preview evaluation completed successfully; a table that
+ failed during a sweep in which it was being previewed shall be previewed again
+ on the next sweep rather than deleting. Disabling and re-enabling retention,
+ restarting the backend, or changing any retention setting shall not produce a
+ second preview for a table that has already completed one.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.5:** When a thresholded table's retained
+ count exceeds that table's configured warning threshold, the system shall emit
+ an operator-visible warning naming the table, its retained count, and the
+ threshold. For `office_routine_runs`, the warning shall also name the routine
+ holding the largest share of retained rows and that share, where the share is
+ that routine's retained rows as a proportion of the table's retained count.
+ When two routines hold an equal largest share, the lower routine identifier
+ shall be named.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.6:** When a table reports remaining
+ backlog after a sweep, the system shall emit an operator-visible warning
+ naming the table, stating that retention is behind, and reporting the number
+ of rows deleted in that sweep. A table that was previewed in that sweep shall
+ not produce this warning however many rows were eligible, because a preview
+ deletes nothing by design and retention is therefore not behind; the preview
+ warning of AC-OFFICE-RUN-HISTORY-RETENTION-003.2 reports its eligible count
+ instead.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.7:** When retention is disabled and a
+ table's retained count exceeds its warning threshold, the system shall emit
+ the same threshold warning and shall additionally state that retention is
+ disabled, so a silent unbounded table is distinguishable from one being
+ actively managed.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.8:** Every warning in this requirement
+ shall be emitted through a channel available in a production build, and shall
+ not be observable only through the debug metrics endpoint.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.9:** A preview sweep's would-delete
+ counts shall be the full eligible count per table, not capped by the batch
+ limit, so an operator is told the real size of what is about to be removed.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.10:** When a swept table's recorded preview
+ state cannot be read or parsed, the system shall treat that table as not yet
+ previewed and shall emit an operator-visible warning naming the table.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-003.11:** Retained counts for the thresholded
+ tables shall be produced by a count evaluation that does not require a sweep
+ to have run, so the counts and the threshold warnings required by AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and
+ AC-OFFICE-RUN-HISTORY-RETENTION-004.8 are available on a fresh install and while retention is disabled. The
+ counts shall be refreshed on the sweep interval and reused between refreshes
+ rather than recomputed for each operator page load or health poll. The first
+ count evaluation shall run at startup, before the delay of
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.10 arms the first sweep, so an operator is
+ not shown countless tables for a sweep interval after every restart. When a
+ count evaluation fails, the system shall keep the counts from the last
+ successful evaluation, report them as stale together with the time they were
+ produced, and emit an operator-visible warning. A failed count shall not
+ suppress the threshold warnings derived from the last successful counts, and
+ shall not fail the sweep. Until the first evaluation has completed, and when it
+ fails with no earlier successful evaluation to fall back on, the system shall
+ report the counts as not yet computed and shall not render them as zero, on the
+ same grounds as AC-OFFICE-RUN-HISTORY-RETENTION-004.7: a count nobody has taken
+ must not read as a table that is empty. Success and failure shall be tracked
+ per thresholded table, so one table's failed count neither discards nor marks
+ stale another table's successful one.
+
+### REQ-OFFICE-RUN-HISTORY-RETENTION-004: Operator configuration and sweep visibility
+
+**Intent:** Retention is configurable, its values are validated, and the result
+of the most recent sweep is readable by an operator without reading logs.
+
+#### Acceptance criteria
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.1:** An operator can read and change
+ whether retention is enabled, the retention window per in-scope table, the
+ sweep interval, the retention floor per in-scope table, the batch limit, and
+ the warning threshold per in-scope table.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.2:** Retention shall be enabled by
+ default. Default values shall be: retention window 30 days for both
+ `office_routine_runs` and `runs`; sweep interval 6 hours; retention floor 50
+ rows per owner for both tables; batch limit 5,000 rows per table per sweep;
+ warning threshold 25,000 rows for `office_routine_runs`, 25,000 for `runs`,
+ and 250,000 for `run_events`.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.3:** The system shall reject a setting
+ outside its permitted range with an error naming the field, and shall leave
+ the stored settings unchanged. Permitted ranges are: retention window 1 to
+ 3,650 days; sweep interval 1 to 168 hours; retention floor 0 to 10,000; batch
+ limit 100 to 100,000; warning threshold 0 or greater, where 0 disables that
+ table's threshold warning.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.4:** When stored retention settings
+ cannot be read or parsed, the system shall use the documented defaults, emit
+ an operator-visible warning, and continue. Unreadable settings shall not
+ disable retention silently and shall not fail startup.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.5:** A settings change shall take effect
+ without a backend restart, and shall apply from the next scheduled sweep
+ rather than interrupting a sweep in progress. Two concurrent writes resolve
+ last-writer-wins; repeating an identical write changes nothing and returns the
+ same normalized document. A sweep shall read the stored settings at its start
+ rather than relying on a value cached when this process last observed a change,
+ so that where several backend processes share one database a process that did
+ not serve the write still sweeps under the new settings rather than the
+ replaced ones. When that read fails or the stored settings cannot be parsed,
+ the sweep shall be skipped and recorded as skipped rather than run against the
+ documented defaults, because a default window is shorter than a window an
+ operator has widened and sweeping under it would delete the history they
+ configured the system to keep. This does not change
+ AC-OFFICE-RUN-HISTORY-RETENTION-004.4, which governs reading settings for
+ reporting and startup, where using the defaults deletes nothing.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.6:** An operator can read, on the System
+ pages, the outcome of the most recent completed sweep: when it ran, which
+ swept tables it previewed, the rows deleted per reported table, the rows it
+ would have deleted per previewed table, the retained count per thresholded
+ table, whether any table has remaining backlog, and any table that failed.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.7:** When no sweep has run since the
+ backend started, the surface in AC-OFFICE-RUN-HISTORY-RETENTION-004.6 shall
+ say so explicitly rather than render an empty or zeroed result that reads as a
+ sweep that deleted nothing.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.8:** The sweep-result surface shall be
+ readable while retention is disabled, reporting retained counts and the
+ disabled state.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-004.9:** A settings write shall replace the whole
+ settings document. A field the caller omits shall take its documented default
+ rather than its previously stored value, so the same request always produces
+ the same stored document. An unrecognized field, a numeric field whose
+ value is not a whole number, a field present with a JSON `null`, and a field
+ whose value is not of its documented type, shall each be rejected with an error
+ naming the field, leaving the stored settings unchanged. `null` shall be
+ rejected rather than treated as omission: omission means the documented
+ default, so reading `null` the same way would let a client silently replace a
+ configured retention window with a shorter default and destroy history the
+ operator meant to keep.
+
+## Out of scope
+
+- **A persisted history of retention sweeps.** The last sweep's result is held
+ in memory and reported alongside the durable warnings. A growing table
+ recording the work of the job that stops tables growing is the same defect in
+ a new place.
+- **Per-workspace or per-routine retention overrides.** Settings are
+ instance-wide. The retention floor is already per owner, which covers what a
+ per-routine override would most often be used for.
+- **Archival or export before deletion.** Deleted history is gone. An install
+ needing it kept sets a longer window or takes a backup.
+- **Anything the deletion policy owns.** Eligibility, ordering, batching,
+ atomicity, and engine parity are specified in
+ [run history retention](run-history-retention.md) and are not restated here.
+
+## Prior art
+
+The receipts for both prior-art legs are recorded once, in
+[run history retention](run-history-retention.md#prior-art). The finding that
+bears on this document specifically: neither surveyed product previews before a
+policy's first deletion, nor warns ahead of the window. Those two behaviors are
+where this capability goes further, and they are the reason this document
+exists rather than being a settings page bolted onto a sweep.
diff --git a/docs/specs/office/requirements/run-history-retention.md b/docs/specs/office/requirements/run-history-retention.md
new file mode 100644
index 00000000000..535a1bf9d04
--- /dev/null
+++ b/docs/specs/office/requirements/run-history-retention.md
@@ -0,0 +1,324 @@
+---
+status: draft
+system: office
+created: 2026-09-09
+owners:
+ - kandev
+---
+
+# Office Run History Retention Requirements
+
+## Overview
+
+Office writes two families of history rows that nothing removes on a schedule.
+`office_routine_runs` records one row per routine firing. `run_events` records
+the timeline of every Office run, append-only, alongside the `runs` queue row
+and its per-run satellites. The only deletions today are manual and structural:
+deleting a routine removes its runs, and deleting a workspace removes everything
+belonging to it. An install that never deletes a routine or a workspace grows
+forever.
+
+The growth is driven by a clock, not by usage. A routine on a `*/5 * * * *`
+schedule fires 105,120 times a year, and each firing writes a routine-run row
+whether or not it does anything. On the reference install, one routine had
+accumulated 323 consecutive `coalesced` rows over 28 days while producing no
+work at all: that is the curve with the loop broken, and a working loop also
+writes the run row, its `run_events` timeline, and its route-attempt rows.
+
+This document bounds those tables. It defines what is history and may be
+deleted, what is live state and must never be deleted on age, when the deletion
+runs, and that it behaves identically on both database engines. What an operator
+is told before the first row is removed, what they are warned about while the
+tables grow, and what they can configure and read is the paired contract in
+[run history retention operations](run-history-retention-operations.md).
+
+Office owns this contract because the rows are defined by Office primitives: the
+routine dispatch ledger, the run queue, and the run event timeline. The System
+pages own the operator surface it reports through. The filesystem and container
+cleanup owned by [storage
+maintenance](../../system-page/requirements/storage-maintenance.md) is a
+separate capability that never touches database rows.
+
+## Terminology
+
+- **Retention sweep** (or **sweep**): one pass that evaluates every in-scope
+ table against the configured policy and deletes the eligible rows.
+- **History row**: a row recording something that already happened, which no
+ live decision reads. History rows are eligible for deletion.
+- **Live-state row**: a row a live decision still reads, regardless of age.
+ Never eligible for age-based deletion.
+- **Recovery-protected run**: a failed `runs` row named by an active
+ `office_agent_pause_recoveries.failed_run_id`; retention keeps it until the
+ recovery row is consumed or discarded.
+- **Run satellite row**: a row keyed by a `runs` row's identifier and owned by
+ it: a `run_events`, `office_run_route_attempts`, or `office_run_skills` entry.
+- **Retention window**: the age past which a history row becomes eligible,
+ measured from the row's own completion time.
+- **Completion time**: `COALESCE(completed_at, created_at)` for a routine-run
+ row and `COALESCE(finished_at, requested_at)` for a `runs` row. Both fallback
+ columns are `NOT NULL`, so completion time is never null.
+- **Retention floor**: most-recent history rows kept per owner regardless of
+ age, so a rarely-firing routine or rarely-woken agent keeps visible history.
+- **Batch limit**: the maximum rows one sweep deletes from one table.
+- **Preview**: a swept table's first evaluation on a database; it counts what it
+ would delete from that table and deletes nothing. Tracked per swept table.
+- **Retained count**: the number of rows a table currently holds — a property of
+ the table, not of a sweep, defined whether or not a sweep has ever run and
+ whether or not retention is enabled.
+- **Swept tables**: the two tables retention selects rows from by policy,
+ `office_routine_runs` and `runs`.
+- **Reported tables**: the five tables a sweep can delete rows from and reports
+ deleted counts for: the two swept tables plus `run_events`,
+ `office_run_route_attempts`, and `office_run_skills`.
+- **Thresholded tables**: the three tables carrying a warning threshold,
+ `office_routine_runs`, `runs`, and `run_events`.
+
+## Requirements
+
+### REQ-OFFICE-RUN-HISTORY-RETENTION-001: History is bounded and live state is not
+
+**Intent:** Bound `office_routine_runs`, `runs`, and the run satellite tables by
+age and by an owner-scoped floor, while guaranteeing that no row another
+decision still reads is removed because it is old.
+
+As an operator running Office continuously, I want old run history removed
+automatically, so a scheduled routine does not grow the database without limit.
+
+#### Acceptance criteria
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.1:** A routine-run row is a history row
+ only when its status is one of `skipped`, `coalesced`, `failed`, `done`, or
+ `cancelled`. When a routine-run row's status is `received` or `task_created`,
+ the system shall treat it as a live-state row and shall not delete it on age,
+ at any age.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.2:** A `runs` row is a history row when its
+ status is `finished`, `failed`, or `cancelled`. `cancelled` is terminal: its
+ only writer moves a row there from `queued` or `claimed` and stamps the
+ completion timestamp in the same statement. When a `runs` row's status is
+ `queued` or `claimed`, the system shall treat it as a live-state row and shall
+ not delete it on age, at any age, including a run parked for a future routing
+ retry. A failed run referenced by an active
+ `office_agent_pause_recoveries.failed_run_id` is also live state for
+ retention and shall remain until that recovery row is consumed or discarded.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.3:** When a history row's completion time is
+ older than that table's retention window, the system shall make it eligible
+ for deletion. Completion time is defined in Terminology. A row's
+ classification as history shall depend on its status alone and shall not
+ additionally require `completed_at` or `finished_at` to be set; a terminal row
+ with an unset stamp is history, dated by its fallback column, and ages out
+ rather than being retained forever.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.4:** The system shall retain the newest
+ history rows of each owner up to that table's retention floor even when they
+ are older than the retention window. The owner is the routine for
+ `office_routine_runs` and the agent profile for `runs`. Newest is completion
+ time descending, with the row identifier descending as the tiebreak.
+ Completion time is never null, so this ordering is total on both engines and
+ does not depend on either engine's default placement of nulls. The identifier
+ gives a stable order for equal timestamps and is not claimed to be
+ chronological.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.5:** When the system deletes a `runs`
+ row, it shall delete that run's satellite rows in the same database
+ transaction, and no satellite row shall remain that references a `runs` row
+ the system has deleted.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.6:** The system shall not delete
+ `run_events` rows for a run it is not deleting in the same transaction, at any
+ age.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.7:** A routine's status shall not exempt
+ its history from retention. When a routine is `paused`, its history rows are
+ evaluated by the same policy as an `active` routine's.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.8:** The system shall not delete, alter,
+ archive, or cancel a task, a session, a task checkout, or a workspace as part
+ of a retention sweep. A routine-run row naming a task in `linked_task_id` may
+ be deleted while that task continues to exist.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.9:** A retained `coalesced` routine-run
+ row may name a `coalesced_into_run_id` whose row has already been deleted.
+ `coalesced_into_run_id` records provenance and is not a referential
+ constraint; the system shall not delete a coalesced row because its target was
+ deleted, and shall not retain a target because a coalesced row names it.
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-001.10:** When a row in a swept table holds a
+ status belonging to neither that table's history set nor its live-state set,
+ the system shall treat it as a live-state row, shall not delete it at any age,
+ and shall emit an operator-visible warning naming the table and the
+ unrecognized status, through the channel required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 in [run history
+ retention operations](run-history-retention-operations.md). The warning shall be
+ produced by the count evaluation required by
+ AC-OFFICE-RUN-HISTORY-RETENTION-003.11 rather than by the deletion path, which
+ cannot observe a status it does not select; it shall therefore be emitted on an
+ install where retention is disabled and no sweep runs. When one table holds more
+ than one unrecognized status, the system shall emit a single warning for that
+ table naming every unrecognized status in ascending lexicographic order with its
+ row count.
+
+### REQ-OFFICE-RUN-HISTORY-RETENTION-002: The retention sweep
+
+**Intent:** Run retention on its own schedule, off the run-claiming hot path, in
+bounded batches, with one sweep at a time and no dependence on database cascade
+behavior.
+
+#### Acceptance criteria
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.1:** The system shall run retention on a
+ dedicated schedule whose interval is configurable in hours, and shall not
+ perform retention work on the Office run-processing tick.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.2:** When a scheduled sweep is due while a
+ previous sweep is still running, the system shall skip the due sweep rather
+ than run two concurrently or queue it, and shall record that it was skipped.
+ Recording a skip shall not replace the reported result of the last completed
+ sweep.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.3:** A sweep shall delete at most the batch
+ limit of rows from `office_routine_runs` and at most that many from `runs`. A
+ preview evaluation shall never be reported as having remaining backlog: it
+ deletes nothing by design, so "retention is behind" is not true of it however
+ many rows were eligible.
+ The limit counts those rows only; every satellite row of a deleted run is
+ removed regardless of the limit. When more rows were eligible than the limit
+ allowed, the system shall complete the sweep, report that table as having
+ remaining backlog, and continue on the next scheduled sweep. Within a table
+ the sweep shall select its batch in completion-time ascending order, with the
+ row identifier ascending as the tiebreak, so the oldest eligible rows are
+ removed first and both engines select the same batch.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.4:** The system shall re-assert every
+ eligibility condition in the deletion itself, not only when selecting
+ candidates. A run that returns to `queued` between selection and deletion,
+ which a scheduled retry does by clearing `finished_at`, shall not be deleted
+ by that sweep. The re-asserted conditions shall include the retention floor as
+ well as status and age.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.5:** Each batch shall be atomic: after a
+ sweep is interrupted by shutdown or error, every batch that was applied is
+ complete, including its satellite rows, and no batch is partially applied.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.6:** Once a table holds no eligible row,
+ a further sweep against unchanged settings shall delete nothing from that table
+ and report zero deletions for it. This criterion is scoped to that drained
+ state and does not contradict AC-OFFICE-RUN-HISTORY-RETENTION-002.3: while a
+ table still reports remaining backlog, the next sweep is required to delete its
+ next batch, so "deletes nothing further" is not claimed of a backlogged table.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.7:** When a table's sweep fails with a
+ database error, the system shall record the failure for that table, continue
+ with the remaining in-scope tables, and retry the failed table on the next
+ scheduled sweep. A retention failure shall not fail backend startup, stop the
+ Office scheduler, or abort the remainder of the sweep. A batch abandoned
+ because its deletion could not be applied consistently shall be recorded as a
+ failure for that table rather than as backlog. Every table whose rows that
+ batch's transaction addressed shall report zero rows deleted for that sweep, so
+ no table reports rows that the rollback restored.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.8:** When retention is disabled, the
+ system shall run no sweep and delete no row, and shall still report retained
+ counts and threshold warnings.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.9:** When every in-scope table is empty
+ or holds no eligible row, the sweep shall complete reporting zero deletions,
+ without error and without a warning.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.10:** The first sweep after the backend
+ starts shall run after a short fixed delay rather than after a full sweep
+ interval, so that an install restarted more often than the interval still runs
+ retention. Later sweeps use the configured interval.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.11:** A sweep shall compute one cutoff
+ instant at its start and evaluate every table against that instant, so two
+ tables in one sweep cannot disagree about what "older than the window" means.
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.12:** On PostgreSQL, where several backend
+ processes can share one database, sweep exclusivity shall hold across
+ processes and not only within one: a backend that cannot acquire the retention
+ lock shall skip its due sweep exactly as it would for a sweep already running
+ in its own process. On SQLite one backend process owns the database file, so
+ the in-process guard is sufficient.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-002.13:** Sweeps shall be scheduled fixed-delay:
+ the next sweep is armed when the previous one finishes, so a sweep running
+ longer than the interval delays its successor rather than causing an immediate
+ second one. A settings change re-arms the delay from the moment of the change.
+ Enabling retention that was disabled arms the next sweep at the same short
+ delay as AC-OFFICE-RUN-HISTORY-RETENTION-002.10 rather than at a full interval.
+ Each backend shall periodically reread the shared settings record, including
+ while retention is disabled, so a backend that did not serve a settings write
+ still adopts enablement and interval changes.
+
+### REQ-OFFICE-RUN-HISTORY-RETENTION-005: Integrity and database engine parity
+
+**Intent:** Retention leaves the database consistent and behaves identically on
+both supported engines, including where the schema differs between them.
+
+#### Acceptance criteria
+
+- **AC-OFFICE-RUN-HISTORY-RETENTION-005.1:** The system shall produce the same
+ observable retention outcome on SQLite and on PostgreSQL for the same settings
+ and the same starting rows: the same rows deleted, the same rows retained, the
+ same reported counts, and the same warnings. This shall hold on the backlog
+ path as well, because batch selection order is fixed by named columns in
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.3 rather than left to the engine.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-005.2:** The system shall delete satellite
+ rows explicitly and shall not depend on a foreign-key cascade to remove them.
+ Neither engine declares a foreign key from a satellite table to `runs`.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-005.3:** After any sequence of sweeps, the
+ concurrency gate that finds a routine's active run by dispatch fingerprint
+ shall return exactly what it would have returned had no sweep run.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-005.4:** After any sequence of sweeps, the
+ lookup that closes out a routine run when its linked task reaches a terminal
+ step shall not resolve a deleted row to a different routine's run. A deleted
+ row resolves to nothing.
+- **AC-OFFICE-RUN-HISTORY-RETENTION-005.5:** Retention shall not change the
+ behavior of deleting a routine or deleting a workspace. Those paths continue
+ to remove every row they remove today, including rows retention has not yet
+ reached.
+
+## Out of scope
+
+Each exclusion below is a decision, not an oversight.
+
+- **Automation run history (`automation_runs` and its tables).** Owned by the
+ automation system, and its published contract is that history is removed only
+ by an explicit per-run or delete-all action. It has the same unbounded-growth
+ gap and needs its own requirement; changing it here would silently break a
+ documented promise.
+- **`office_activity_log`, inbox dismissals, and approval rows.** The
+ [inbox requirement](inbox.md) already states these accumulate indefinitely and
+ excludes activity-log retention. Reopening it belongs to that contract.
+- **`office_cost_events`.** Deleting cost rows changes reported spend and budget
+ enforcement: a decision about financial records, not about storage.
+- **Detecting a stuck routine.** The 323-row reference case came from a routine
+ that fired correctly and did nothing useful for 28 days. Bounding its history
+ does not detect it; a detector for that state is a scheduler or stall
+ visibility concern.
+- **Reclaiming file bytes after deletion.** Deleting rows does not shrink a
+ SQLite file. `VACUUM` is already an operator action on the System pages and a
+ sweep does not trigger it.
+- **Archival or export before deletion.** Deleted history is gone. An install
+ needing it kept sets a longer window or takes a backup.
+- **Per-workspace or per-routine retention overrides.** Settings are
+ instance-wide here. The retention floor is already per owner, which covers what
+ a per-routine override would most often be used for.
+- **A persisted history of retention sweeps.** The last sweep's result is held
+ in memory and reported alongside the durable warnings. A growing table
+ recording the work of the job that stops tables growing is the same defect in
+ a new place.
+- **Tables outside Office.** Nothing here changes task, session, workflow,
+ plugin, or auth storage.
+
+## Prior art
+
+Receipts for both legs. The design document carries what each finding changed.
+
+**Our own wiki: not consulted, tool unavailable.** The `@henry` pin resolved
+`~/.obsidian-wiki/config` to `config.henry`, giving
+`OBSIDIAN_VAULT_PATH=/Users/henry/Documents/henry/wiki` and
+`QMD_WIKI_COLLECTION=wiki`. Neither retrieval path ran: `qmd` and
+`obsidian-wiki` are absent from `PATH`, no QMD MCP tool is registered in this
+session, and the vault directory returns `Operation not permitted` both
+sandboxed and unsandboxed, a macOS file-access restriction on this process
+rather than a missing vault. The grep fallback is blocked the same way, so this
+leg is a tooling gap, not evidence the wiki is silent on retention.
+
+**What other products shipped: consulted.** Queried the `saas-kb` server
+(`search_fsm_docs`, `category: "ai_sdlc"`) three times: run-history retention and
+database growth; session history retention and automatic deletion; and
+scheduled-automation run-history limits. Relevance was low across all three,
+itself a finding about corpus coverage. Two useful hits: **GitLab Duo** deletes
+sessions 30 days after last activity, which anchors the default window here; and
+**the Claude apps gateway** documents four tables with per-table windows
+enforced by one hourly sweep, marking one table "until deleted via the API"
+instead of giving it a window, which is the split this document draws between
+history rows and live-state rows.
+
+Neither previewed before a policy's first deletion, nor warned ahead of the
+window. Those are where this capability goes further, because an upgrade that
+silently deletes a year of history on its first sweep is the failure mode a
+shipped default carries.
diff --git a/docs/specs/office/system-design/run-history-retention-operations.md b/docs/specs/office/system-design/run-history-retention-operations.md
new file mode 100644
index 00000000000..f4195860f7c
--- /dev/null
+++ b/docs/specs/office/system-design/run-history-retention-operations.md
@@ -0,0 +1,379 @@
+---
+status: current
+system: office
+requirements:
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-003
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-004
+---
+
+# Office Run History Retention Operations System Design
+
+## Purpose and boundaries
+
+This design covers the operator half of run history retention: the settings
+record, the per-table preview marker, the reporting value, the health check, and
+the read/write System page surface. The sweep itself, the eligibility
+predicates, the batching, and the engine-parity guarantees are designed in [run
+history retention](run-history-retention.md).
+
+The two documents share one component, `internal/office/retention`, and one
+scheduler goroutine. They are split because they are two contracts with two
+consumers: a scheduler deleting rows, and an operator reading a page. The split
+also keeps each document inside the specification size limit without cutting
+either contract.
+
+Adjacent contracts read and constrained but not owned:
+
+- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health`
+ response the System page's health card renders.
+- `internal/system/settings.Store` — the key/value settings table, already used
+ by storage maintenance under one JSON key with normalization on read.
+
+## Component: the operator surface
+
+### Settings
+
+One `settings` key, `office_run_retention`, holding a JSON document,
+read through a `Get`/`Save` pair with `Normalize` on both, matching
+`internal/system/storage/settings.go`. Unparseable content returns the defaults
+plus a sentinel error the caller turns into a health issue, never a boot failure
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.4).
+
+```json
+{
+ "enabled": true,
+ "sweep_interval_hours": 6,
+ "batch_limit": 5000,
+ "routine_runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 },
+ "runs": { "window_days": 30, "floor_per_owner": 50, "warn_rows": 25000 },
+ "run_events": { "warn_rows": 250000 }
+}
+```
+
+`run_events` carries only a threshold: its lifetime is its run's, so it has no
+window and no floor of its own. Ranges and rejection behavior are
+AC-OFFICE-RUN-HISTORY-RETENTION-004.2 and .3; validation returns a field-named
+error and writes nothing, as `validateRange` does for storage maintenance.
+Writes are last-writer-wins through `Store.Save`; `CompareAndSwap` is not used
+because these are operator-scale settings edited from one page, and an identical
+repeated write is indistinguishable from no write
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.5).
+
+**Each sweep re-reads the settings from the store at its start, and skips if that
+read fails.** The buffered
+wake channel that re-arms the interval is process-local, so on a deployment where
+several backends share one PostgreSQL database — the deployment
+AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists for — it only ever reaches the
+process that served the `PUT`. A backend relying on a cached effective value
+could win the sweep lock still holding the policy the operator has just replaced
+and delete rows the new policy retains. Re-reading costs one indexed key lookup
+per sweep interval, against the risk of deleting history under a window the
+operator already widened. The wake channel keeps its job — re-arming the timer
+promptly in the process that saw the change — and stops being the only path by
+which a change reaches a sweep.
+
+A failed or unparseable re-read **skips the sweep** rather than falling back to
+the documented defaults. Everywhere else in this design an unreadable settings
+document yields the defaults and a health issue
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.4), which is right for reporting and for
+startup because neither deletes anything. It is wrong here: the default window is
+30 days, an operator who widened theirs to 3,650 has by definition configured
+something longer, and falling back would delete the decade of history they
+configured the system to keep. Reading settings to *show* them fails open;
+reading them to *delete by* fails closed.
+
+
+### The preview marker
+
+A second `settings` key, `office_run_retention_preview_completed`,
+holding a JSON object keyed by swept table name whose values are the timestamp
+at which that table's preview completed:
+`{"office_routine_runs": "...", "runs": "..."}`.
+
+The marker is **per table, not per database**. A single global flag is unsafe
+against AC-OFFICE-RUN-HISTORY-RETENTION-002.7, which lets one table fail while
+the sweep as a whole still completes: if `runs` errored during the preview sweep
+and `office_routine_runs` succeeded, a global flag would be written anyway and
+`runs` would delete for real on the next sweep having never shown an operator a
+would-delete count — the exact failure this capability exists to prevent. With a
+per-table marker, `runs` is simply previewed again next sweep
+(AC-OFFICE-RUN-HISTORY-RETENTION-003.4).
+
+A table's entry is written only when that table's preview evaluation completed
+successfully, is never cleared by a settings change or a restart, and is not
+written by a deleting sweep. A table whose preview finds nothing still gets its
+entry, so its next sweep deletes normally.
+
+If the key is present but unparseable, every swept table is treated as **not yet
+previewed** and `office_retention_preview_unreadable` is raised
+(AC-OFFICE-RUN-HISTORY-RETENTION-003.10). The asymmetry is deliberate: a
+spurious re-preview deletes nothing and costs one sweep, whereas assuming a
+preview had completed permits a first deletion no operator ever saw. The safe
+direction is the one that cannot delete.
+
+The preview's per-table counts are uncapped by the batch limit
+(AC-OFFICE-RUN-HISTORY-RETENTION-003.9): capping them would report 5,000 to an
+operator holding 200,000 eligible rows, which is exactly the number the warning
+exists to convey.
+
+
+### Reporting
+
+An in-memory `LastSweep` value replaced wholesale at the end of each sweep that
+actually ran: start and finish times, and per **reported** table the deleted
+count, a backlog flag, and an error string. Deleted counts are counted from
+**committed** transactions only: when a `runs` batch is abandoned and rolled back
+(AC-OFFICE-RUN-HISTORY-RETENTION-002.7), the three satellite tables whose rows
+that transaction addressed report **zero** deleted for that sweep, not the counts
+their statements returned before the rollback. The failure is recorded against
+`runs`, but the satellites must not show rows the rollback restored — the one
+surface an operator has for "what actually happened" would otherwise be wrong
+precisely in the failure case it exists for. Per **swept** table it also carries
+whether that table was previewed in this sweep and, when it was, its
+`WouldDelete` count — without that field the preview's headline numbers, which
+AC-OFFICE-RUN-HISTORY-RETENTION-003.2 and
+AC-OFFICE-RUN-HISTORY-RETENTION-003.9 require an operator to see, would have
+nowhere to be read from. Because the preview marker is per table, `preview` is a
+per-table flag rather than one flag for the sweep. Skips are held separately and
+never overwrite this value. Nothing is persisted
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.6, and the "no persisted sweep history"
+exclusion: a growing table recording the work of the job that stops tables
+growing is the same defect in a new place). Before the first sweep the value is
+absent, and the surface says so rather than rendering zeros
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.7).
+
+**Retained counts do not come from `LastSweep`.** They are a property of the
+table, not of a sweep, and three ACs need them when no sweep has run at all:
+AC-OFFICE-RUN-HISTORY-RETENTION-003.7 (threshold warning while retention is
+disabled), AC-OFFICE-RUN-HISTORY-RETENTION-004.8 (surface readable while
+disabled), and AC-OFFICE-RUN-HISTORY-RETENTION-002.8 (disabled means no sweep,
+yet counts are still reported). A separate `RetainedCounts` value therefore
+holds one count per thresholded table — produced by the census described next —
+plus, for `office_routine_runs`, the top routine by retained rows for
+AC-OFFICE-RUN-HISTORY-RETENTION-003.5's attribution.
+
+**The count evaluation is a status census, not a bare `COUNT(*)`.** For each
+swept table it issues one
+`SELECT status, COUNT(*) FROM
GROUP BY status`. The retained count is the
+sum of those rows, so the census costs one scan rather than two, and the same
+result set is what detects a status in neither the history nor the live-state set
+and raises `office_retention_unknown_status:
`
+(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). That detector has to live here rather
+than in the sweep: the sweep's predicate selects `status IN ()`
+and therefore structurally cannot observe a status it does not select.
+`run_events` is thresholded but not swept and has no status column, so it keeps a
+plain `COUNT(*)`.
+
+It is refreshed on the sweep interval by the same scheduler goroutine — on that
+schedule whether or not retention is enabled, since a disabled install is
+precisely the one whose tables grow unattended, and is also the only install
+where an unrecognized status would otherwise never be noticed — and served from
+memory in between.
+
+**The first evaluation runs at `Start`, not at the first sweep.** The first sweep
+is deliberately delayed five minutes (AC-OFFICE-RUN-HISTORY-RETENTION-002.10);
+hanging the first count off that tick would leave every restart with five minutes
+of absent counts, and a fresh install with none at all until it had swept once —
+while AC-OFFICE-RUN-HISTORY-RETENTION-003.7 and -004.8 require the threshold
+warning and the surface to work on exactly those installs.
+
+It runs **on the scheduler goroutine, not on the caller of `Start`**. The census
+is three unbounded scans of the largest tables in the database, and the install
+that most needs them is the one where they take longest; blocking `Start` on them
+would make bounding these tables a cause of slow boots. `Start` returns
+immediately, the tables read *not yet computed* until the first census lands
+seconds later, and that state is already required and already renders honestly.
+
+`RetainedCounts` is therefore **tri-state per thresholded table**, not a number
+that defaults to zero: *not yet computed*, *fresh as of T*, or *stale as of T*.
+Zero is a real measurement and must not be how "nobody has counted yet" renders —
+the same argument AC-OFFICE-RUN-HISTORY-RETENTION-004.7 makes for `LastSweep`,
+and the reason AC-OFFICE-RUN-HISTORY-RETENTION-003.11 states it. The three states
+are tracked **per table**, so one table's failing query neither discards nor
+staleness-marks another's successful one; a failure with no prior success leaves
+that table at *not yet computed* rather than fabricating a zero.
+
+The evaluation is explicitly **not** computed per health poll or per page load:
+`/api/v1/system/health` is polled by every open browser tab, and issuing three
+unbounded `COUNT(*)`s against the largest tables in the database on each poll
+would make this feature a cause of the load it exists to prevent
+(AC-OFFICE-RUN-HISTORY-RETENTION-003.11).
+
+Three surfaces, in descending durability:
+
+1. **Structured logs.** One `info` line per sweep. One `warn` line per condition
+ in REQ-OFFICE-RUN-HISTORY-RETENTION-003, carrying the table, the counts, and
+ the threshold. Always on, in every profile.
+2. **Health issues.** The package implements `health.Checker` with
+ `Name() = "Office run retention"` and `Category() = "office"`, returning a
+ `health.Issue` per active condition with
+ `FixURL = "/settings/system/data-storage"`, which is the live route
+ registered in `apps/web/src/settings-routes.tsx` — note the suffix, as
+ `/settings/system/data` is not a registered path and `/settings/system/database`
+ is only a redirect to it. `internal/health/checks_test.go` pins fix URLs to
+ live routes ("expectedGitHubFixURL is the live route in ..."), and this one
+ is pinned the same way. This is the production-visible surface
+ required by AC-OFFICE-RUN-HISTORY-RETENTION-003.8 and is why the debug
+ metrics endpoint is not it: `/debug/vars` is gated on the dev profile.
+ Issue ids are stable and one per condition:
+ `office_retention_preview_pending`, `office_retention_backlog:
`,
+ `office_retention_preview_unreadable`, and
+ `office_retention_unknown_status:
`.
+ `Check` returns issues sorted by id, matching `storage.Runtime.Check`, so the
+ health card's order is stable across polls.
+ For `office_routine_runs`, the threshold issue's message names the routine
+ holding the largest share of retained rows and that share
+ (AC-OFFICE-RUN-HISTORY-RETENTION-003.5) — attribution computed from the same
+ retained-count query, not a second detector.
+3. **expvar.** Counters under `office_retention_*` following
+ `office/scheduler/metrics_vars.go`. Development convenience only; nothing in
+ REQ-OFFICE-RUN-HISTORY-RETENTION-003 depends on it.
+
+
+### HTTP and frontend
+
+`GET /api/v1/system/retention` returns effective settings, `LastSweep`, the
+separately-held skip record, and `RetainedCounts`.
+
+`PUT /api/v1/system/retention` **replaces the whole document**; it is not a
+merge patch. A field the caller omits takes its documented default rather than
+its stored value, so the same request body always yields the same stored
+document and a repeated identical write is genuinely a no-op — which is what
+AC-OFFICE-RUN-HISTORY-RETENTION-004.5's "identical repeated write changes
+nothing and returns the same normalized document" requires. A merge would break
+that: under a merge, whether a request is a no-op depends on what was stored
+before it. An unrecognized field, a numeric field whose value is not a whole
+number, a field present with a JSON `null`, and a field whose value is of the
+wrong type, are each rejected with a field-named 400 and nothing is written
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.9); rejecting rather than ignoring an
+unknown field means a client that misspells `window_days` is told so instead of
+silently getting the default. `null` needs saying because this endpoint is a full
+replace: since an omitted field deliberately means "take the default", the
+tempting reading of `null` is the same one, and that reading turns
+`{"runs": {"window_days": null}}` into a silent 3,650-to-30-day reduction that
+destroys a decade of history on the next sweep. Decode into pointer fields and
+reject an explicit null, rather than into value fields where `null` and omission
+are indistinguishable. Successful writes return the normalized document.
+
+Both routes are admin-scoped like the other System routes and are readable while
+retention is disabled (AC-OFFICE-RUN-HISTORY-RETENTION-004.8).
+
+One card on **Settings > System > Data & Logs**, the page served at
+`/settings/system/data-storage` and rendered by
+`apps/web/components/settings/system/data-logs-settings.tsx`, beside
+`database-stats-card.tsx`: the enable toggle, the numeric fields, and the
+last-sweep readout. It follows the storage-maintenance cards' shape. All new
+copy goes through `t()` and must ship in `pt-pt`, `zh-cn`, `zh-hk`, and `zh-tw`;
+`pnpm run i18n:check` and the new-code ratchet gate the build. Health issue
+titles and messages stay English, matching every other backend-produced
+`health.Issue`.
+
+
+## Ordering, concurrency, and failure
+
+| Question | Answer | AC |
+|---|---|---|
+| Concurrent settings writes | Last-writer-wins; an identical repeat changes nothing | 004.5 |
+| Settings unreadable | Defaults used, health issue raised, sweep proceeds | 004.4 |
+| Settings out of range | Rejected with the field named, stored settings unchanged | 004.3 |
+| Retention disabled | No sweep, no deletion; counts and threshold warnings still reported | 002.8, 003.7, 004.8 |
+| Preview scope | Per swept table, not per database; a table that failed its preview is previewed again | 003.4 |
+| Preview marker unreadable | Treated as not previewed; re-preview deletes nothing | 003.10 |
+| Retained counts | Table property from a `GROUP BY status` census; first evaluation at `Start`, then on the sweep interval; served from memory; computed even while disabled | 003.11 |
+| Counts before the first evaluation, or first evaluation fails | Reported *not yet computed* per table, never as zero; state tracked per thresholded table | 003.11, 004.7 |
+| Unknown status detected | By the census, not the sweep predicate; one issue per table listing every unrecognized status in ascending order | 001.10 |
+| Preview on a table with more eligible rows than the batch limit | Reports the full eligible count; never reported as backlog | 003.6, 003.9 |
+| Settings changed on another backend | Each sweep re-reads settings at its start, so a process that did not serve the write still uses the new policy | 004.5 |
+| Settings unreadable at sweep start | Sweep skipped and recorded as skipped; never run under the shorter default window | 004.5, 004.4 |
+| Batch abandoned and rolled back | Satellite tables report zero deleted, not their pre-rollback statement counts | 002.7, 004.6 |
+| Settings write shape | Full replace; omitted field takes its default; unknown field 400 | 004.9 |
+
+## Testing
+
+Unit tests in `internal/office/retention`, plus the frontend checks below.
+
+- The first sweep on a seeded database deletes nothing and reports a
+ would-delete count; the second deletes (003.1, 003.4).
+- A preview on a database with more eligible rows than the batch limit reports
+ the full eligible count (003.9).
+- A table that errors during the sweep in which it was being previewed is
+ previewed again on the next sweep instead of deleting, while a sibling table
+ that succeeded proceeds to delete (003.4).
+- A preview marker that is present but unparseable causes a re-preview, not a
+ deletion, and raises `office_retention_preview_unreadable` (003.10).
+- Retained counts and the threshold warning are produced on a fresh install
+ before any sweep, and while retention is disabled (003.7, 003.11, 004.8), and
+ a health poll does not issue a table count.
+- Counts are available immediately after `Start`, without advancing any clock to
+ the first sweep's delay (003.11). A test that waits out the delay would pass
+ against an implementation that hangs the first count off the sweep tick, which
+ is the defect this asserts against.
+- Before the first evaluation, and when the first evaluation fails with no
+ earlier success, the surface reports the table as *not yet computed* and not as
+ zero (003.11, 004.7).
+- With one thresholded table's count query failing and the other two succeeding,
+ the two keep fresh counts and only the failing one is marked, and the threshold
+ warnings derived from the successful counts still fire (003.11).
+- A swept table holding a status in neither status set raises
+ `office_retention_unknown_status:
` from the census while retention is
+ **disabled** and no sweep has ever run (001.10, 003.11).
+- A preview on a table with more eligible rows than the batch limit reports the
+ full eligible count and raises no backlog warning (003.6, 003.9).
+- A settings change written through one store handle is used by a sweep driven
+ from a second handle that never saw the change notification (004.5).
+- A backend census refresh adopts a settings change written by another backend,
+ including when retention was disabled, and arms the appropriate sweep timer
+ (002.13, 004.5).
+- A sweep whose settings read fails is skipped and recorded as skipped, and
+ deletes nothing under the default window (004.5).
+- A `runs` batch abandoned after its retry reports zero deleted for the three
+ satellite tables rather than their pre-rollback counts (002.7, 004.6).
+- A recorded skip leaves the previous `LastSweep` readable and unchanged (002.2,
+ 004.7).
+- A settings write omitting a field stores that field's default; a write with an
+ unknown field or a fractional number is rejected 400 naming the field and
+ stores nothing; the same write applied twice is a no-op (004.9).
+
+Commands:
+
+```
+cd apps/backend
+go test ./internal/office/... -race -count=1
+KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... -count=1
+cd apps/web && pnpm run typecheck && pnpm run i18n:check
+```
+
+## Rejected alternatives
+
+- **One global "preview completed" flag.** Simpler, and unsafe: because
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.7 lets one table fail while the sweep
+ completes, a global flag is written even when a table never got its preview,
+ and that table then deletes for real having shown the operator nothing. The
+ per-table marker costs one JSON object.
+- **Compute retained counts inside the health check.** Direct, and it puts three
+ unbounded `COUNT(*)`s on a route every open browser tab polls. Refreshing on
+ the sweep interval and serving from memory gives the same number without
+ making the bound-the-tables feature a source of load on those tables.
+- **A merge-patch settings write.** Under a merge, whether a request is a no-op
+ depends on what was stored before it, which
+ AC-OFFICE-RUN-HISTORY-RETENTION-004.5 forbids.
+- **Deletion off by default.** Safe, and it means the gap stays open on every
+ install that never visits the settings page. The per-table preview gives the
+ same protection without that outcome.
+- **Report warnings only through `/debug/vars`.** That endpoint is gated on the
+ dev profile, so on a production build the warnings would not exist
+ (AC-OFFICE-RUN-HISTORY-RETENTION-003.8).
+
+## Prior art, applied
+
+`internal/system/storage` supplies the settings storage and normalization
+pattern, the hours-based interval with min and max bounds, and the
+`health.Checker` route to a production-visible warning; `storage.Runtime.Check`
+also sorts its issues by id, which this check matches so the health card's order
+is stable across polls.
+
+Neither GitLab Duo nor the Claude apps gateway previews before a policy's first
+deletion or warns ahead of the window. Those are this capability's additions,
+and they are the whole reason this document is separate from the sweep's.
diff --git a/docs/specs/office/system-design/run-history-retention.md b/docs/specs/office/system-design/run-history-retention.md
new file mode 100644
index 00000000000..c0831c0ad1d
--- /dev/null
+++ b/docs/specs/office/system-design/run-history-retention.md
@@ -0,0 +1,571 @@
+---
+status: current
+system: office
+requirements:
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-001
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-002
+ - REQ-OFFICE-RUN-HISTORY-RETENTION-005
+---
+
+# Office Run History Retention System Design
+
+## Purpose and boundaries
+
+This design adds one scheduled sweep that deletes aged Office run history from
+five tables. It changes no run lifecycle, no routine dispatch decision, and no
+task, session, or checkout.
+
+Office owns the sweep because the eligibility rules are defined by Office
+primitives. The settings record, the health check, the reporting value, and the
+System page surface are the operator half of the same capability and are
+designed in [run history retention
+operations](run-history-retention-operations.md), which is where every
+requirement in REQ-OFFICE-RUN-HISTORY-RETENTION-003 and -004 is satisfied.
+
+Adjacent contracts read and constrained but not owned:
+
+- `internal/health` — `Checker`, `Issue`, and the `/api/v1/system/health`
+ response the System page's health card renders.
+- `internal/system/settings.Store` — the key/value settings table, already used
+ by storage maintenance under one JSON key with normalization on read.
+- `internal/runs/repository/sqlite` — the `runs` queue and `run_events` access
+ methods.
+- `internal/office/repository/sqlite` — the schema owner for `runs`,
+ `run_events`, `office_run_route_attempts`, `office_run_skills`, and
+ `office_routine_runs`.
+
+## Measured starting state
+
+Read from the reference install's SQLite database on 2026-09-09. These numbers
+set the defaults and are the baseline any regression test can be written
+against.
+
+| Table | Rows | Notes |
+|---|---|---|
+| `office_routine_runs` | 326 | 323 `coalesced`, 3 `task_created` |
+| `runs` | 53 | |
+| `run_events` | 340 | across 53 runs, mean 6.4 per run |
+| `office_run_route_attempts` | 55 | |
+| `office_run_skills` | 382 | |
+
+The 323 `coalesced` rows span 2026-08-03 to 2026-08-31 and belong to a single
+routine that is now `paused`. Orphan `run_events` today: zero, because nothing
+has ever deleted a `runs` row.
+
+## Schema facts this design depends on
+
+Verified by reading the schema owners, not assumed.
+
+- `office_routine_runs` (`office/repository/sqlite/base.go`) has
+ `FOREIGN KEY (routine_id) REFERENCES office_routines(id) ON DELETE CASCADE`
+ on both engines, and SQLite opens with `_foreign_keys=on`
+ (`internal/db/sqlite.go`). Its two indexes are partial and serve the dispatch
+ gate, not an age scan: `idx_office_routine_runs_active_fingerprint`
+ (`WHERE status = 'task_created'`) and `idx_office_routine_runs_linked_task`
+ (`WHERE linked_task_id != ''`).
+- `run_events`, `office_run_route_attempts`, and `office_run_skills` declare
+ **no foreign key to `runs`** on either engine. Confirmed in
+ `office/repository/sqlite/base.go` and in the PostgreSQL conformance snapshot
+ `internal/persistence/storeconformance/testdata/upgrades/v0.93.0/postgres.sql`,
+ where `run_events` has only a primary key and one index. A cascade would
+ therefore delete nothing; satellite deletion must be explicit
+ (AC-OFFICE-RUN-HISTORY-RETENTION-005.2).
+- `run_events.seq` is assigned by `AppendRunEvent` as
+ `COALESCE(MAX(seq) + 1, 0)` scoped to the run. Deleting a live run's whole
+ timeline restarts its sequence at zero, and the run detail view's incremental
+ tail reads `WHERE seq > afterSeq`. This is why
+ AC-OFFICE-RUN-HISTORY-RETENTION-001.6 forbids event deletion outside the
+ transaction that deletes the run.
+- `runs` indexes are `idx_run_status_requested (status, requested_at)` and the
+ partial unique `idx_run_idempotency`. Nothing indexes `finished_at`.
+- `office_agent_pause_recoveries.failed_run_id` identifies failed runs that an
+ active pause recovery still needs. It has no foreign key, so retention uses a
+ correlated `NOT EXISTS` predicate and the supporting
+ `idx_office_agent_pause_recoveries_failed_run` index.
+- `ScheduleRetry` (`runs/repository/sqlite/runs.go`) sets
+ `status = 'queued', finished_at = NULL` on an existing run, keyed by id with
+ **no status guard in its `WHERE` clause**. Any terminal run can therefore
+ become live again at any moment, which is the race
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.4 closes. Because the guard is absent, a
+ `cancelled` row is resurrectible on exactly the same terms as a `failed` one,
+ so admitting `cancelled` to the history set adds no new race — it is covered
+ by the same re-assertion.
+- `CancelRunsWhere` (`runs/repository/sqlite/cancel.go`) is documented as "the
+ single writer of the terminal cancel state on the runs table". It sets
+ `status = 'cancelled', cancel_reason = ?, finished_at = ?` and is guarded by
+ `status IN ('queued', 'claimed')`, so a cancelled row is always terminal and
+ always carries a completion timestamp. It is production-reachable through
+ `CancelRunsForTasks` from `office/service/tree_controls.go` and
+ `office/repository/sqlite/participants.go`. It writes the literal string, so
+ `office/models/enums.go`'s four-value `RunStatus` block does not enumerate it:
+ the database has five `runs` statuses, not four. This is why
+ AC-OFFICE-RUN-HISTORY-RETENTION-001.2 classifies `cancelled` as history and
+ AC-OFFICE-RUN-HISTORY-RETENTION-001.10 makes any sixth value fail safe.
+- Every production writer of a terminal `runs` status stamps the completion
+ timestamp: `FinishRun` sets `finished_at = now`, `MarkRunFailed` sets
+ `finished_at = COALESCE(finished_at, now)`, and `CancelRunsWhere` sets it
+ outright. Only the test helper `SetRunStatusForTest` can leave a terminal row
+ with a null `finished_at`. The `COALESCE(finished_at, requested_at)` fallback
+ in AC-OFFICE-RUN-HISTORY-RETENTION-001.3 is therefore defensive rather than a
+ routine path — but it is also what makes the ordering key non-null, which is
+ what keeps the floor deterministic across engines (see below).
+- `runs.requested_at` and `office_routine_runs.created_at` are both
+ `TIMESTAMP NOT NULL`, so the completion-time expression is total. This matters
+ for parity, not just tidiness: SQLite and PostgreSQL differ in where they sort
+ nulls by default, so an `ORDER BY` over a nullable timestamp would rank the
+ floor differently on the two engines from identical data.
+- `CleanExpired` (`runs/repository/sqlite/runs.go`) already deletes terminal
+ `runs` rows older than a cutoff and **has no production caller** — only
+ tests reach it. It deletes only `runs`, so wiring it as-is would orphan every
+ satellite row. It is superseded by the batched, satellite-aware delete below
+ rather than reused.
+
+## Component: `internal/office/retention`
+
+One package holding policy, settings, the sweep, and the health check.
+
+### Ownership and lifecycle
+
+A single goroutine owner modelled on `internal/system/storage.Scheduler`: a
+`Start(ctx)` that is a no-op when already running, a `Stop()` that cancels and
+joins, and a buffered wake channel so a settings change re-arms the interval
+without interrupting a sweep in progress
+(AC-OFFICE-RUN-HISTORY-RETENTION-004.5). `Stop` is joined from the same place
+that stops the Office scheduler.
+
+The loop is deliberately not the Office run-processing tick
+(`office/service/scheduler_integration.go`, `DefaultTickInterval = 5s`). That
+tick already carries `RecoverStale` and `ReapStaleCheckouts` unthrottled and is
+the run-claim hot path; on SQLite a bulk delete there contends with the single
+writer that claims runs, 17,280 times a day, for an input that changes on a
+scale of days (AC-OFFICE-RUN-HISTORY-RETENTION-002.1).
+
+A `sweeping bool` guarded by the same mutex makes a due sweep a skip rather than
+a second goroutine (AC-OFFICE-RUN-HISTORY-RETENTION-002.2). That guard is
+process-local, which is sufficient on SQLite, where the database file is owned
+by one backend process. On PostgreSQL, where several backends can share one
+database, it is not: two backends would each see `sweeping == false` and sweep
+the same rows concurrently.
+
+The sweep therefore also takes a PostgreSQL advisory lock, and it must be a
+**session-scoped, non-blocking** one, which is a different shape from every
+existing advisory lock in this repository:
+
+```
+conn := db.Conn(ctx) // one dedicated connection
+SELECT pg_try_advisory_lock(:retention_key) // boolean, returns immediately
+... whole sweep, every table, every batch, on the pool ...
+SELECT pg_advisory_unlock(:retention_key) // on that same connection
+conn.Close() // deferred
+```
+
+Both properties are load-bearing and neither is optional:
+
+- **Session-scoped, not transaction-scoped.** The sweep is multi-transaction by
+ construction: AC-OFFICE-RUN-HISTORY-RETENTION-002.5 makes each batch its own
+ transaction, and AC-OFFICE-RUN-HISTORY-RETENTION-002.7 requires one table's
+ failure to leave another table's committed deletions intact, which forbids
+ wrapping the sweep in a single transaction. A `pg_advisory_xact_lock` is
+ released when its transaction ends, so it would protect one batch and then let
+ a second backend in between tables — the interleaving
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.12 exists to prevent. The lock is held on a
+ connection checked out for the sweep and released in a `defer`; the batches
+ themselves continue to use the pool.
+- **`try`, not the blocking form.** AC-OFFICE-RUN-HISTORY-RETENTION-002.12 says a
+ backend that cannot acquire the lock *skips*. `pg_advisory_xact_lock` and
+ `pg_advisory_lock` wait instead of failing, which would convert a concurrent
+ sweep into a queued one and eventually stall the scheduler behind a long sweep.
+ `pg_try_advisory_lock` returns `false` immediately; on `false` the backend
+ records a skip and returns, exactly as for a local concurrent sweep.
+
+This deliberately departs from the established repo pattern
+`SELECT pg_advisory_xact_lock(hashtextextended(?, 0))`
+(`office/repository/sqlite/participants.go`, `internal/secrets/sqlite_store.go`,
+`internal/workflow/repository/phase2_sqlite.go`), and the departure is the point:
+each of those call sites performs its entire unit of work inside the one
+transaction that holds the lock, and each *wants* to wait rather than skip. The
+sweep can do neither. The key is derived the same way, `hashtextextended` over a
+constant distinct from every key those sites use, so retention never contends
+with participant-seat, secret-transfer, or workflow-phase locking.
+
+**If the lock connection drops mid-sweep**, PostgreSQL releases the session's
+advisory locks as part of ending the session, so a crashed or partitioned backend
+cannot wedge retention permanently — that self-healing is the reason for a
+session lock rather than a lease row in the `settings` table, which would need its
+own expiry and its own stale-holder rule. The cost is that the surviving sweep no
+longer holds exclusivity without knowing it, so the sweep **verifies the lock
+connection is still alive between tables** and, if it is not, stops before the
+next table, records that table as skipped rather than failed, and returns. It
+does not attempt to re-acquire mid-sweep: a re-acquisition after another backend
+has taken the lock would produce exactly the concurrent sweep this protects
+against. Batches already committed stay committed, which
+AC-OFFICE-RUN-HISTORY-RETENTION-002.5 permits.
+
+A skip is recorded as its own value — a skip counter and a last-skip timestamp —
+and does **not** overwrite `LastSweep`. `LastSweep` holds the last sweep that
+actually ran, so a burst of skips cannot blank the operator's view of the last
+real result (AC-OFFICE-RUN-HISTORY-RETENTION-002.2, and
+AC-OFFICE-RUN-HISTORY-RETENTION-004.7, which forbids rendering a
+never-swept-looking surface when a sweep has in fact run).
+
+The first sweep after `Start` is armed at a fixed 5-minute delay rather than a
+full interval (AC-OFFICE-RUN-HISTORY-RETENTION-002.10). Arming at the interval,
+as the storage scheduler does with its 24-hour default, means an install
+restarted more often than the interval never sweeps at all; 5 minutes keeps
+startup clear of schema init and run recovery without depending on uptime. Each
+sweep computes `time.Now().UTC()` once and passes that instant to every table,
+so two tables in one sweep cannot disagree about the cutoff
+(AC-OFFICE-RUN-HISTORY-RETENTION-002.11).
+
+Scheduling is **fixed-delay, not fixed-rate**: the next sweep is armed when the
+previous one returns, so a sweep that overruns its interval delays its successor
+instead of causing an immediate second one (which the concurrency guard would
+only skip anyway, turning a slow sweep into a stream of skips). A settings
+change re-arms the delay from the moment of the change rather than from the last
+sweep, and enabling retention that was disabled arms at the same 5-minute delay
+as a fresh start rather than at a full interval — otherwise an operator who
+enables retention on a 168-hour interval waits a week to see whether it works
+(AC-OFFICE-RUN-HISTORY-RETENTION-002.13).
+
+The census timer also re-reads shared settings. It runs while retention is
+disabled, so other backends discover enablement and interval changes and re-arm
+their timers.
+
+### Eligibility, expressed once
+
+Two predicates, each defined in exactly one place and reused by the count, the
+preview, and the delete.
+
+**Routine runs.** History statuses are `skipped`, `coalesced`, `failed`, `done`,
+`cancelled`. `received` and `task_created` are absent by construction, which is
+what makes AC-OFFICE-RUN-HISTORY-RETENTION-005.3 hold: the dispatch gate
+`GetActiveRunForFingerprint` reads only `status = 'task_created'`, so no sweep
+can change its answer.
+
+```
+DELETE FROM office_routine_runs
+WHERE id IN (
+ SELECT id FROM (
+ SELECT id,
+ ROW_NUMBER() OVER (
+ PARTITION BY routine_id
+ ORDER BY COALESCE(completed_at, created_at) DESC, id DESC
+ ) AS rn
+ FROM office_routine_runs
+ WHERE status IN ()
+ ) ranked
+ WHERE rn > :floor
+ AND completion_time < :cutoff
+ ORDER BY completion_time ASC, id ASC
+ LIMIT :batch
+)
+AND status IN ()
+AND COALESCE(completed_at, created_at) < :cutoff
+```
+
+where the ranked subquery also projects
+`COALESCE(completed_at, created_at) AS completion_time`.
+
+Two orderings appear here and they are not the same ordering; conflating them is
+the defect this section exists to prevent.
+
+- The **`ORDER BY` inside `ROW_NUMBER()`** ranks rows *within* a partition so the
+ floor keeps the newest per owner: `completion_time DESC, id DESC`
+ (AC-OFFICE-RUN-HISTORY-RETENTION-001.4).
+- The **`ORDER BY` on the outer select** decides *which* eligible rows a
+ batch-limited sweep takes: `completion_time ASC, id ASC`, oldest first
+ (AC-OFFICE-RUN-HISTORY-RETENTION-002.3). The window function's ordering does
+ not reach the outer `LIMIT`, so without this clause the engine is free to
+ return any subset and the two engines may drain a backlog differently from
+ identical data — which would contradict
+ AC-OFFICE-RUN-HISTORY-RETENTION-005.1 and make the parity test below flake for
+ a reason unrelated to a genuine engine difference.
+
+`id` is a UUID and so is not chronological; it is used only as a total, stable
+tiebreak for equal timestamps, in both orderings.
+
+The trailing `AND` clauses are the re-assertion required by
+AC-OFFICE-RUN-HISTORY-RETENTION-002.4. For this table the whole ranked subquery
+is re-evaluated inside the `DELETE`, so the floor is re-asserted atomically along
+with status and age; the two-phase `runs` path below has to do that explicitly.
+
+Window functions are available on both engines: SQLite 3.54.0 through
+`github.com/mattn/go-sqlite3 v1.14.33`, verified by running this exact
+`ROW_NUMBER() OVER (PARTITION BY routine_id ...)` against the reference
+database.
+
+**Runs.** History is `status IN ('finished','failed','cancelled')`, partitioned
+by `agent_profile_id`, ranked `COALESCE(finished_at, requested_at) DESC, id DESC`
+and batch-ordered `COALESCE(finished_at, requested_at) ASC, id ASC`. A failed
+run named by an active `office_agent_pause_recoveries.failed_run_id` is
+protected by a `NOT EXISTS` clause in this predicate. Count, selection, and
+delete use the same clause, so a preview cannot promise deletion of a run that
+the recovery flow still needs. There is no
+`finished_at IS NOT NULL` conjunct: requiring one would make a terminal row with
+an unset stamp immortal and unobservable, which
+AC-OFFICE-RUN-HISTORY-RETENTION-001.3 forbids. `queued` and `claimed` are
+absent, which covers a routing-parked run whose `earliest_retry_at` is far in
+the future (AC-OFFICE-RUN-HISTORY-RETENTION-001.2).
+
+A status in neither set is treated as live state and raises
+`office_retention_unknown_status:
`
+(AC-OFFICE-RUN-HISTORY-RETENTION-001.10). Both sets are closed and asserted
+against `office/models/enums.go` plus the literal-SQL writers in a test, so a
+sixth status cannot enter the database without failing that test — the failure
+mode that hid `cancelled` in the first place.
+
+**That warning needs a producer, and the eligibility predicate cannot be it.**
+The predicate selects `status IN ()`, so a row holding an
+unrecognized status is never selected, never counted, and never seen: a fail-safe
+whose only detector is a CI test fires on the developer's machine and stays
+silent on the install that actually has the row. The producer is instead the
+**status census** — one
+
+```
+SELECT status, COUNT(*) FROM GROUP BY status
+```
+
+per swept table, run by the retained-count evaluation described in [run history
+retention operations](run-history-retention-operations.md#reporting). Three
+consequences follow from siting it there rather than in the sweep, and each one
+closes a hole:
+
+- The census **subsumes the retained count** rather than adding a second scan:
+ the retained count is the sum of the census rows, so one `GROUP BY` yields
+ both. (`run_events` is thresholded but not swept and has no status column, so
+ it keeps a plain `COUNT(*)`.)
+- It runs **whether or not retention is enabled**, because the count evaluation
+ does (AC-OFFICE-RUN-HISTORY-RETENTION-003.11). A disabled install is precisely
+ where an unrecognized status would otherwise never be noticed, since
+ AC-OFFICE-RUN-HISTORY-RETENTION-002.8 means no sweep runs there at all.
+- It sees a status **because the row exists**, not because the row was
+ selectable, which is the property the eligibility predicate structurally
+ cannot have.
+
+One issue per table, not one per status: a table with several unrecognized
+statuses raises the single id `office_retention_unknown_status:
` whose
+message lists every unrecognized status **in ascending lexicographic order** with
+its row count. Ordering is named because the message is compared across health
+polls; an unordered list would make a stable condition look like a changing one.
+
+### Deleting a run
+
+Per batch, one transaction, satellites first, run last:
+
+1. Select up to `batch_limit` eligible run ids, excluding runs named by an
+ active `office_agent_pause_recoveries.failed_run_id`.
+2. `DELETE FROM run_events WHERE run_id IN (...)`
+3. `DELETE FROM office_run_route_attempts WHERE run_id IN (...)`
+4. `DELETE FROM office_run_skills WHERE run_id IN (...)`
+5. `DELETE FROM runs WHERE id IN (:selected_ids) AND id IN ()` — the re-assertion is the **whole ranked subquery**, not just the
+ status and cutoff conjuncts, so the retention floor is re-evaluated at delete
+ time along with them. Unlike the single-statement `office_routine_runs`
+ delete, this path selected its ids in a separate earlier statement, so a
+ `ScheduleRetry` in between can re-rank a partition and push a row that was
+ `rn > floor` at selection to `rn <= floor` now. Re-asserting only status and
+ age would delete a row that has since become floor-protected
+ (AC-OFFICE-RUN-HISTORY-RETENTION-002.4).
+
+Step 5 can delete fewer rows than steps 2 to 4 addressed, when a `ScheduleRetry`
+resurrected a run between selection and delete. That is the correct outcome for
+AC-OFFICE-RUN-HISTORY-RETENTION-002.4 only if the whole batch rolls back rather
+than leaving a live run without its timeline. **The transaction is therefore
+rolled back and retried once with a fresh selection when the step 5 row count
+does not match the selected id count**; a second mismatch rolls back again and
+abandons the batch.
+
+An abandoned batch is recorded as a **failure** for that table, raising
+`office_retention_failed:
`, not as backlog
+(AC-OFFICE-RUN-HISTORY-RETENTION-002.7). The distinction is not cosmetic:
+backlog means work correctly deferred by the batch limit and is expected on a
+large install, so routing this case there would file the one genuinely dangerous
+outcome under the one routine one. The table is retried on the next scheduled
+sweep either way, but only the failure classification tells an operator that a
+deletion could not be applied consistently.
+
+Step 5 can only ever delete *fewer* rows than steps 2 to 4 addressed, never
+more, because it is bounded by the same selected id set. This is the one place
+where the naive implementation silently corrupts a live run, and it is the
+reason satellite deletion is not a separate statement outside the transaction.
+
+`run_events` is never addressed by any predicate other than membership in this
+id set (AC-OFFICE-RUN-HISTORY-RETENTION-001.6). There is no age-based delete on
+`run_events`.
+
+### Indexes to add
+
+Retention adds indexes that serve the sweep.
+
+- `idx_office_routine_runs_retention ON office_routine_runs(routine_id, status, (COALESCE(completed_at, created_at)) DESC, id DESC)`
+- `idx_runs_retention ON runs(agent_profile_id, status, (COALESCE(finished_at, requested_at)) DESC, id DESC)`
+- `idx_office_agent_pause_recoveries_failed_run ON office_agent_pause_recoveries(failed_run_id)`
+
+Added through the existing `office/repository/sqlite` schema path so both the
+fresh-install `CREATE` and the upgrade path get them, and recorded in the
+conformance fixtures.
+
+## Ordering, concurrency, and failure
+
+| Question | Answer | AC |
+|---|---|---|
+| Sweep ordering across tables | `office_routine_runs`, then `runs` with its satellites. Independent sets; the order is fixed only so results and logs are reproducible. | 002.6 |
+| Floor ordering | `COALESCE(completed_at, created_at) DESC, id DESC` / `COALESCE(finished_at, requested_at) DESC, id DESC` | 001.4 |
+| Two sweeps due at once | Second is skipped, not queued, and recorded as skipped | 002.2 |
+| Sweep vs. live writer | Selection is a snapshot; status, age **and floor** are re-asserted in the delete; a batch whose step 5 count disagrees is rolled back | 002.4, 002.5 |
+| Sweep interrupted | Each batch is one transaction; applied batches are whole, the interrupted one is not applied | 002.5 |
+| First sweep after startup | Armed at a fixed 5-minute delay, not a full interval | 002.10 |
+| Cutoff instant | Computed once per sweep, shared by every table | 002.11 |
+| Re-run once no eligible rows remain | Deletes nothing further, reports zero. While backlog remains, the next sweep deletes the next batch by 002.3 — the two are not in tension because 002.6 is scoped to the drained state | 002.6, 002.3 |
+| One table errors | That table is recorded failed; remaining tables continue; retried next sweep; startup unaffected | 002.7 |
+| Empty tables | Zero deletions, no error, no warning | 002.9 |
+| Deleted coalesce target | Allowed; `coalesced_into_run_id` is provenance, read by nothing | 001.9 |
+| Routine paused | Irrelevant to eligibility | 001.7 |
+| Cancelled run | History, like `finished`/`failed`; its writer stamps the completion timestamp | 001.2 |
+| Terminal row, null completion stamp | Still history; dated by `created_at` / `requested_at` | 001.3 |
+| Status in neither set | Treated as live state, never deleted, warned | 001.10 |
+| Which rows a batch-limited sweep takes | `completion_time ASC, id ASC` — oldest first, named columns | 002.3, 005.1 |
+| Batch abandoned after retry | Recorded as that table's failure, not as backlog | 002.7 |
+| Preview finds more eligible rows than the batch limit | Not backlog. A preview deletes nothing by design, so "retention is behind" would be false; it reports the full eligible count instead | 002.3, 003.6, 003.9 |
+| Sweep skipped | Recorded separately; never overwrites the last real `LastSweep` | 002.2, 004.7 |
+| Two backends, one PostgreSQL | Session-scoped `pg_try_advisory_lock` held on a dedicated connection for the whole sweep; the loser skips without waiting | 002.12 |
+| Lock connection drops mid-sweep | PostgreSQL releases the lock with the session; the sweep stops before the next table and records it skipped, and does not re-acquire | 002.12 |
+| Scheduling model | Fixed-delay from the end of the previous sweep; a settings change re-arms from the change | 002.13 |
+
+## Testing
+
+Unit and repository tests in `internal/office/retention` and
+`internal/office/repository/sqlite`, plus the persistence gates.
+
+Behaviors that must have a test, because each is a way the naive implementation
+is wrong:
+
+- A `task_created` routine run older than any window survives, and
+ `GetActiveRunForFingerprint` still finds it (001.1, 005.3).
+- A `queued` run with `finished_at` cleared by `ScheduleRetry` survives (001.2).
+- A run resurrected between selection and delete leaves both the run and its
+ full `run_events` timeline intact (002.4, and the rollback above).
+- After a sweep, no `run_events`, `office_run_route_attempts`, or
+ `office_run_skills` row references a missing `runs` row (001.5, 005.2).
+- No `run_events` row of a surviving run is ever deleted (001.6).
+- A routine with 3 history rows all older than the window keeps all 3 under a
+ floor of 50; a routine with 200 keeps exactly 50 (001.4).
+
+- Two sweeps triggered concurrently produce one sweep and one recorded skip
+ (002.2).
+- A sweep hitting the batch limit reports backlog and the next sweep continues,
+ and its satellite rows are deleted in full rather than capped (002.3, 003.6).
+
+- A terminal row whose completion timestamp is unset is dated by `created_at`
+ or `requested_at` and ages out rather than being retained forever (001.3).
+ This test is only satisfiable because AC-OFFICE-RUN-HISTORY-RETENTION-001.2
+ classifies history by status alone; an implementation that also required
+ `finished_at IS NOT NULL` would retain the row forever and fail here.
+- A `cancelled` run older than the window is deleted together with its satellite
+ rows, and a `cancelled` run inside the floor is retained (001.2). Seed it
+ through the real cancel path, not by writing the status directly, so the test
+ fails if that path stops stamping the completion timestamp.
+- A `runs` row holding a status in neither the history set nor the live-state set
+ survives every sweep at any age and raises
+ `office_retention_unknown_status:runs` (001.10). Three companions: the same row
+ raises the same issue with retention **disabled**, where no sweep runs at all;
+ a table holding two unrecognized statuses raises **one** issue listing both in
+ ascending order with their counts; and a test asserts the two status sets
+ together cover every value in `office/models/enums.go` *and* every status
+ literal written by SQL in `internal/runs/repository/sqlite`, so a future sixth
+ value cannot be added silently — the exact gap through which `cancelled` was
+ missed.
+- With more eligible rows than the batch limit, the sweep deletes the oldest
+ eligible rows first, and the same seed data yields the identical deleted set on
+ SQLite and PostgreSQL (002.3, 005.1). Without the outer `ORDER BY` this test is
+ the one that fails.
+- A row that becomes floor-protected between selection and delete is not deleted
+ by the `runs` path (002.4).
+
+- A batch abandoned after its retry is reported as that table's failure and not
+ as backlog (002.7).
+
+- Two backends against one PostgreSQL run one sweep, and the loser records a
+ skip (002.12). The seed must give the winner **more than one table** to sweep,
+ so that a transaction-scoped lock — which would release between tables and let
+ the loser in — fails this test rather than passing it. A companion test drops
+ the winner's lock connection mid-sweep and asserts it stops before the next
+ table and does not re-acquire.
+- A `runs` batch abandoned after its retry reports zero rows deleted for
+ `run_events`, `office_run_route_attempts` and `office_run_skills`, not the
+ counts the rolled-back statements addressed (002.7, 004.6).
+
+Commands:
+
+```
+cd apps/backend
+go test ./internal/office/... ./internal/runs/... -race -count=1
+go run ./cmd/sqlguard ./internal
+go test -race ./internal/persistence/storeconformance -count=1
+KANDEV_TEST_POSTGRES_DSN= go test -race ./internal/office/... \
+ ./internal/persistence/storeconformance -count=1
+cd apps/web && pnpm run typecheck && pnpm run i18n:check
+```
+
+The PostgreSQL line is not optional. The repository's dialect-sensitive suites
+self-skip when `KANDEV_TEST_POSTGRES_DSN` is unset, so a green local run without
+it proves nothing about AC-OFFICE-RUN-HISTORY-RETENTION-005.1. The engine-parity
+test asserts identical deleted sets, retained sets, and counts from identical
+seed data on both engines.
+
+## Rejected alternatives
+
+- **Wire the existing `CleanExpired` and stop.** One line, and it orphans every
+ `run_events` row it passes, permanently and invisibly, because no foreign key
+ on either engine would clean up after it.
+- **Add `ON DELETE CASCADE` to the satellite tables instead.** A schema change
+ to three tables on two engines, requiring a table rebuild on SQLite, to avoid
+ three `DELETE` statements. It also hides the deletion from the code that has
+ to count it for the sweep report.
+- **Age-prune `run_events` directly.** The obvious implementation, and the one
+ that breaks the run detail view's incremental tail and can restart a live
+ run's sequence at zero. Forbidden by 001.6.
+- **Run the sweep on the 5s Office tick.** Rejected in 002.1.
+- **Count-only retention, keep newest N per owner with no window.** Bounds the
+ table but makes "how long is my history kept" unanswerable for an install
+ with mixed routine frequencies. The floor covers the case count-only is good
+ at; the window covers the case it is bad at.
+- **Deletion off by default.** Safe, and it means the gap stays open on every
+ install that never visits the settings page. The per-table preview, designed
+ in [run history retention
+ operations](run-history-retention-operations.md), gives the same protection
+ without that outcome.
+- **Persist sweep history.** Named in the operations requirement's exclusions.
+
+- **Rely on the window function's `ORDER BY` to order the batch.** It ranks rows
+ inside each partition; it does not order the rows the outer `LIMIT` draws
+ from. Leaving the outer select unordered makes a backlog sweep's deleted set
+ engine-dependent and quietly breaks
+ AC-OFFICE-RUN-HISTORY-RETENTION-005.1 only on installs large enough to exceed
+ the batch limit — the installs that need this feature most.
+
+## Prior art, applied
+
+**Wiki: unavailable** — receipt in the requirement document. Nothing here should
+be read as departing from a wiki position, because none could be consulted.
+
+**`internal/automation/run_retention.go`** is the closest in-repo precedent, and it
+prunes worktrees rather than rows. Three things are taken from it: retention scoped
+per owner rather than globally, so one noisy owner cannot evict a quiet one's
+only record; a bounded sweep window so a backlog drains across sweeps instead
+of walking the whole table each time; and re-checking liveness immediately
+before the destructive act, which appears here as the re-asserted predicate and
+the batch rollback. One thing is deliberately not taken: it hangs its sweep off
+a finalization hook, which ties cleanup frequency to firing frequency and leaves
+a stopped routine's history untouched forever. This design uses a clock.
+
+**`internal/system/storage`** supplies the scheduler shape, the settings
+storage and normalization pattern, the hours-based interval with min and max
+bounds, and the `health.Checker` route to a production-visible warning.
+
+**GitLab Duo** and **the Claude apps gateway** are surveyed in the requirement
+document's Prior art. What this design takes from them: the 30-day default
+window, and the history-vs-live-state split (their per-table windows with one
+table marked "until deleted via the API" rather than given a window). Neither
+previews before the first deletion; that addition is ours, and it exists because
+this ships enabled by default onto installs that already hold history.