-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdb.js
More file actions
75 lines (67 loc) · 2.83 KB
/
Copy pathdb.js
File metadata and controls
75 lines (67 loc) · 2.83 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
/**
* WHAT THIS FILE DOES
*
* Owns the database. Two things live here:
*
* 1. `pool` — a shared pool of Postgres connections, reused by every request
* instead of opening a new connection each time.
* 2. `initDb()` — creates the pgvector extension and the documents table if
* they do not exist yet. Safe to run on every boot.
*
* The documents table is the whole storage layer:
*
* id a unique id per chunk
* content the chunk's text, sent to the LLM as context
* source the filename, so answers can cite where they came from
* embedding the chunk's vector, what the similarity search compares
*
* See docs/04-code-walkthrough.md for a line-by-line explanation.
*
* @author Raqibul Hasan Moon <rhmoon21@gmail.com>
* @created 2026-08-12
*/
const { Pool } = require("pg");
const pool = new Pool({
connectionString: process.env.DATABASE_URL,
});
/* The VECTOR(3072) dimension matches the output of Gemini's gemini-embedding-001 model exactly. If you use a different embedding model in the future, check its output dimensions and update this number to match.
*/
/*
* Scaling caveat for 3072 dimensions.
*
* pgvector allows a `vector` column up to 16,000 dimensions, but both index
* types (HNSW and IVFFlat) only support up to 2,000. At 3072 this table
* therefore CANNOT carry a vector index — every /chat query runs an exact
* sequential scan over all rows.
*
* That is fine, and even preferable, at the scale this project targets: exact
* search returns true nearest neighbours where an approximate index only
* estimates them. It stops being fine somewhere in the tens of thousands of
* chunks, when the scan starts dominating query latency.
*
* The escape hatches at that point, in increasing order of effort:
* 1. Store as `halfvec(3072)` and index with HNSW — halfvec indexes support
* up to 4,000 dimensions. Costs some precision, keeps the model.
* 2. Ask Gemini for a smaller embedding via `outputDimensionality` (768 and
* 1536 are supported) and index normally. Requires re-embedding everything.
*
* Either way, changing the dimension means every stored vector must be
* regenerated — old and new embeddings are not comparable.
*/
// Runs on every boot. Both statements are IF NOT EXISTS, so this is safe to
// repeat — it doubles as the "migration" for a project this size.
async function initDb() {
// Requires an image that ships pgvector (see docker-compose.yml). The stock
// postgres image fails here with "extension \"vector\" is not available".
await pool.query(`CREATE EXTENSION IF NOT EXISTS vector`);
await pool.query(`
CREATE TABLE IF NOT EXISTS documents (
id UUID PRIMARY KEY,
content TEXT NOT NULL,
source TEXT NOT NULL,
embedding VECTOR(3072)
)
`);
console.log("Database ready");
}
module.exports = { pool, initDb };