Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
71 changes: 71 additions & 0 deletions .agents/skills/performance/SKILL.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,71 @@
---
name: performance
description: Performance rules and known hot paths for this codebase. Use before touching impl/query_*.ml, impl/transact.ml, impl/db.ml, impl/storage.ml, impl/pull_api.ml, impl/entity.ml, or any code that loops over datoms, query rows, or index contents.
---

# Performance

Measured profile (large Logseq graphs): queries returning 30k–70k rows dominate;
GC ~26%, List ops ~10%. The rules below keep new code off the quadratic paths and
keep hot paths allocation-light.

## Index access discipline

- Never walk a whole index (`datoms db Eavt ()`, unbounded `Avet ~a` slices) or
materialize `entity`/`ent_of_id` inside loops when a bounded seek suffices.
Prefer `~e`/`~a`/`~v`-constrained `datoms` queries and datom-level checks.
- Whole-index work belongs only in inherently whole-db operations: `diff`,
export, serialize/restore, GC (`collect_garbage`), checksum/validate.
- `index_datoms_seq`/`reverse_index_datoms_seq` stream without materializing;
`visible_index_datoms`/`raw_index_datoms_list` copy the full index — use the
seq variants inside loops.

## List anti-patterns banned on hot paths

Per-datom, per-row, per-block loops must not contain:

- `xs @ [x]`, `List.append`, `List.concat` inside folds/loops — O(n²). Cons onto
a reversed accumulator, or use `Hashtbl`-keyed groups / `Queue` / `Rrbvec`,
then `List.rev`/`concat` once at the end.
- `List.mem` / `List.assoc` / `List.find` / `List.exists` / `List.nth` inside a
loop over a growing or large collection — build a `Hashtbl`/Set once, then
O(1) lookups.
- `List.length` recomputed per iteration — hoist it or thread a counter.
- `List.filter`/`map` chains that re-traverse large data when one fold suffices.
- `List.sort`/`List.sort_uniq` where a `Hashtbl` dedup or a min-fold applies
(e.g. sorting only to take `List.hd`).
- `List.combine`/`List.split`/`Array.to_list`/`List.of_seq`/`List.of_array` per
row — keep arrays positional, hash the key→index map once.
- Non-tail recursion over unbounded lists (stack depth risk at 100k+).

Cheap-constant exceptions: fixed small lists like `Schema.schema_fields`
(10 elements) — `List.mem` on them is fine.

## Known hot paths (audit Oct 2026)

- `impl/query_where.ml` — join/scan loops; `relation.rows` is
`query_result array list` with typed-hash dedup already — keep it that way.
- `impl/query_api.ml:295` — `collect_find_specs` + `List.combine`/`Array.to_list`
per result row; precompute a `Hashtbl attr→index` per query instead.
- `impl/db.ml` — `diff` (whole-DB; pairwise scans now `(e,a)`→values
Hashtbls), `eavt_datoms`, `remove_facts_from_duplicate_tables`/
`refresh_indexes_with_tx_data` (retract rebuilds now batched per tx —
keep them batched), `find_active_datom_by_fact` (min-fold, keep it).
- `impl/transact.ml:754` — `dedupe_facts` now Hashtbl buckets on
upsert-merge; order must stay first-occurrence.
- `impl/storage.ml` — `collect_garbage` (live set is a Hashtbl),
`tail_datom_count` (concat-to-count).
- `impl/entity.ml:48` — `raw_forward_entity_attrs` assoc/remove_assoc grouping
per datom → O(D²) on wide entities.
- `impl/pull_api.ml:492/513/604` — `List.mem` ancestry check per recursion level.
- `impl/conn.ml:150,178` — `storage_tail @ [report.tx_data]` per tx.

## Checklist before submitting

- For each `List.` call inside a loop: what is the input size at this call site,
and who reaches it from a public endpoint (`transact`, `q`, `pull`, `entity`,
`datoms`, `diff`, storage GC)?
- Whole-db scans only in diff/export/restore/GC/checksum.
- Allocations: avoid `map|>filter|>map` chains and intermediate lists on
>1k-item data; fuse into a single fold.
- Verify no `Seq` was forced into a `List` where streaming suffices.
111 changes: 79 additions & 32 deletions impl/db.ml
Original file line number Diff line number Diff line change
Expand Up @@ -208,7 +208,12 @@ let find_active_datom_by_fact db datom =
in
match PSet.slice ~from_:bound ~to_:bound ~cmp db.eavt_index @ duplicate_matches with
| [] -> None
| matches -> Some (matches |> List.sort (Util.compare_datom Eavt) |> List.hd)
| first :: rest ->
Some
(List.fold_left
(fun best d -> if Util.compare_datom Eavt d best < 0 then d else best)
first
rest)

let add_datom_to_indexes db datom =
{ db with
Expand All @@ -222,49 +227,69 @@ let add_datom_to_indexes db datom =
; max_datom_e = max db.max_datom_e datom.e
}

let remove_fact_from_duplicate_tables db active =
let remove_facts_from_duplicate_tables db actives =
(* Retraction removes the fact entirely, so every stored copy — including
the ones kept out of the PSet indexes in the duplicate tables — must go.
The tables are shared with prior db values, so rebuild them rather than
mutate in place. *)
let duplicate_datoms = List.filter (fun d -> not (same_fact d active)) db.duplicate_datoms in
if duplicate_datoms == db.duplicate_datoms then
mutate in place. Facts are batched so the whole duplicate list is
filtered and re-sorted once per transaction, not once per datom. *)
if db.duplicate_datoms = [] then
db
else
let duplicate_aevt_datoms = List.sort (Util.compare_datom Aevt) duplicate_datoms in
let duplicate_avet_datoms =
duplicate_datoms
|> List.filter (fun datom -> Schema.schema_attr_is_avet_accessible db.schema datom.a)
|> List.sort (Util.compare_datom Avet)
let removed : (int * string, value list) Hashtbl.t =
Hashtbl.create (List.length actives)
in
{ db with
duplicate_datoms
; duplicate_aevt_datoms
; duplicate_avet_datoms
; duplicate_eavt_by_entity = duplicate_eavt_by_entity duplicate_datoms
; duplicate_aevt_by_attr = duplicate_datoms_by_attr duplicate_aevt_datoms
; duplicate_avet_by_attr = duplicate_datoms_by_attr duplicate_avet_datoms
}
List.iter
(fun (d : datom) ->
let key = (d.e, d.a) in
match Hashtbl.find_opt removed key with
| Some vs -> Hashtbl.replace removed key (d.v :: vs)
| None -> Hashtbl.replace removed key [ d.v ])
actives;
let is_removed (d : datom) =
match Hashtbl.find_opt removed (d.e, d.a) with
| Some vs -> List.exists (value_equal d.v) vs
| None -> false
in
let duplicate_datoms = List.filter (fun d -> not (is_removed d)) db.duplicate_datoms in
if duplicate_datoms == db.duplicate_datoms then
db
else
let duplicate_aevt_datoms = List.sort (Util.compare_datom Aevt) duplicate_datoms in
let duplicate_avet_datoms =
duplicate_datoms
|> List.filter (fun datom -> Schema.schema_attr_is_avet_accessible db.schema datom.a)
|> List.sort (Util.compare_datom Avet)
in
{ db with
duplicate_datoms
; duplicate_aevt_datoms
; duplicate_avet_datoms
; duplicate_eavt_by_entity = duplicate_eavt_by_entity duplicate_datoms
; duplicate_aevt_by_attr = duplicate_datoms_by_attr duplicate_aevt_datoms
; duplicate_avet_by_attr = duplicate_datoms_by_attr duplicate_avet_datoms
}

let refresh_indexes_with_tx_data db tx_data =
let db =
let db, removed_actives =
List.fold_left
(fun db datom ->
(fun (db, removed) datom ->
if datom.added then
add_datom_to_indexes db datom
add_datom_to_indexes db datom, removed
else
match find_active_datom_by_fact db datom with
| None -> db
| None -> db, removed
| Some active ->
let db = remove_fact_from_duplicate_tables db active in
{ db with
eavt_index = PSet.remove active db.eavt_index
; aevt_index = PSet.remove active db.aevt_index
; avet_index = PSet.remove active db.avet_index
})
db
( { db with
eavt_index = PSet.remove active db.eavt_index
; aevt_index = PSet.remove active db.aevt_index
; avet_index = PSet.remove active db.avet_index
}
, active :: removed ))
(db, [])
tx_data
in
let db = remove_facts_from_duplicate_tables db removed_actives in
invalidate_attr_tables db

let with_datoms db datoms =
Expand Down Expand Up @@ -994,9 +1019,31 @@ let index_range context db attr ?start ?stop () =
let diff left right =
let left_datoms = visible_index_datoms left Eavt in
let right_datoms = visible_index_datoms right Eavt in
( List.filter (fun d -> not (List.exists (same_fact d) right_datoms)) left_datoms
, List.filter (fun d -> not (List.exists (same_fact d) left_datoms)) right_datoms
, List.filter (fun d -> List.exists (same_fact d) right_datoms) left_datoms
(* (e, a) -> values hash table per side turns the pairwise same_fact
scan into an O(1)-amortized lookup per datom. *)
let fact_table datoms =
let tbl : (int * string, value list) Hashtbl.t =
Hashtbl.create (List.length datoms)
in
List.iter
(fun (d : datom) ->
let key = (d.e, d.a) in
match Hashtbl.find_opt tbl key with
| Some vs -> Hashtbl.replace tbl key (d.v :: vs)
| None -> Hashtbl.replace tbl key [ d.v ])
datoms;
tbl
in
let has_fact tbl (d : datom) =
match Hashtbl.find_opt tbl (d.e, d.a) with
| Some vs -> List.exists (value_equal d.v) vs
| None -> false
in
let left_facts = fact_table left_datoms in
let right_facts = fact_table right_datoms in
( List.filter (fun d -> not (has_fact right_facts d)) left_datoms
, List.filter (fun d -> not (has_fact left_facts d)) right_datoms
, List.filter (fun d -> has_fact right_facts d) left_datoms
)

let squuid_counter = ref 0
Expand Down
5 changes: 3 additions & 2 deletions impl/storage.ml
Original file line number Diff line number Diff line change
Expand Up @@ -403,7 +403,8 @@ let settings (db : db) =
]

let collect_garbage storage =
let live = storage_root_addresses storage in
let live = Hashtbl.create 257 in
List.iter (fun address -> Hashtbl.replace live address ()) (storage_root_addresses storage);
storage.storage_list_addresses ()
|> List.filter (fun address -> not (List.mem address live))
|> List.filter (fun address -> not (Hashtbl.mem live address))
|> storage.storage_delete
22 changes: 17 additions & 5 deletions impl/transact.ml
Original file line number Diff line number Diff line change
Expand Up @@ -752,11 +752,23 @@ let apply_tx context tx_ops db =
|> List.filter (fun datom -> datom.e <> old_e)
in
let dedupe_facts datoms =
datoms
|> List.fold_left
(fun deduped d ->
if List.exists (context.same_fact d) deduped then deduped else d :: deduped)
[]
let buckets : (entity_id * attr, datom list) Hashtbl.t =
Hashtbl.create (List.length datoms)
in
List.filter
(fun d ->
let key = (d.e, d.a) in
match Hashtbl.find_opt buckets key with
| Some bucket when List.exists (context.same_fact d) bucket -> false
| Some bucket ->
Hashtbl.replace buckets key (d :: bucket);
true
| None ->
Hashtbl.replace buckets key [ d ];
true)
datoms
(* the old fold accumulated reversed — keep that output order *)
|> List.rev
in
let remapped_ref_datoms =
referring_datoms
Expand Down
Loading