-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathingest.js
More file actions
80 lines (67 loc) · 2.83 KB
/
Copy pathingest.js
File metadata and controls
80 lines (67 loc) · 2.83 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
/**
* WHAT THIS FILE DOES
*
* The ingestion pipeline. Turns an uploaded PDF into rows in the database:
*
* PDF bytes -> plain text -> overlapping chunks -> vectors -> documents table
*
* The chunking is the interesting part. A whole document is too big to embed
* usefully, so it is cut into ~500 character pieces that overlap slightly, and
* each piece gets its own vector. Retrieval later works on these pieces.
*
* See docs/04-code-walkthrough.md for a line-by-line explanation.
*
* @author Raqibul Hasan Moon <rhmoon21@gmail.com>
* @created 2026-08-12
*/
const { PDFParse } = require("pdf-parse");
const { v4: uuidv4 } = require("uuid");
const { pool } = require("./db");
const { embedText } = require("./embeddings");
/*
500 characters per chunk, with 50 characters of overlap between neighbours.
Why the overlap? Without it, a sentence that straddles a boundary gets split, half in one chunk, half in the next — and neither piece makes sense on its own when retrieved. The overlap keeps those boundary sentences intact.
For most technical docs, 500 is a good starting point. Dense legal or financial text tends to need something closer to 300.
*/
function chunkText(text, chunkSize = 500, overlap = 50) {
const chunks = [];
let start = 0;
while (start < text.length) {
const end = Math.min(start + chunkSize, text.length);
chunks.push(text.slice(start, end).trim());
start += chunkSize - overlap;
}
return chunks.filter((chunk) => chunk.length > 50);
}
async function ingestDocument(buffer, filename) {
// pdf-parse v2 replaced the v1 callable export with this class. The parser
// holds a PDF.js worker open, so destroy() must run even if parsing throws —
// otherwise the process keeps a live worker per failed upload.
const parser = new PDFParse({ data: buffer });
let text;
try {
({ text } = await parser.getText());
} finally {
await parser.destroy();
}
const chunks = chunkText(text);
console.log(`Processing ${chunks.length} chunks from "${filename}"`);
// Sequential on purpose. Promise.all() over the chunks would fire every
// embedding request at once and trip Gemini's per-minute rate limit on any
// sizeable document. Slower, but it finishes.
//
// Each chunk is also its own transaction, so a failure midway leaves the
// earlier chunks committed and the document partially ingested. Acceptable
// while ingestion is manual; if this ever runs unattended, wrap the loop in a
// single transaction or delete by `source` before re-ingesting.
for (const chunk of chunks) {
const embedding = await embedText(chunk);
await pool.query(
`INSERT INTO documents (id, content, source, embedding)
VALUES ($1, $2, $3, $4::vector)`,
[uuidv4(), chunk, filename, JSON.stringify(embedding)],
);
}
return chunks.length;
}
module.exports = { ingestDocument };