-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathquery.js
More file actions
76 lines (65 loc) · 3.39 KB
/
Copy pathquery.js
File metadata and controls
76 lines (65 loc) · 3.39 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
/**
* WHAT THIS FILE DOES
*
* The query pipeline — the "R" and the "G" in RAG.
*
* question -> vector -> find 5 nearest chunks -> send them to the LLM
*
* Retrieval (R) happens in SQL: pgvector compares the question's vector to
* every stored chunk vector and returns the closest ones. Generation (G) is the
* LLM writing an answer from ONLY those chunks — which is what keeps the answer
* grounded in your documents instead of the model's general knowledge.
*
* See docs/04-code-walkthrough.md for a line-by-line explanation.
*
* @author Raqibul Hasan Moon <rhmoon21@gmail.com>
* @created 2026-08-12
*/
const { pool } = require("./db");
const { embedText, generateAnswer } = require("./embeddings");
/*
The <=> operator is pgvector's cosine distance. Semantically similar text produces vectors that point in the same direction — so the distance between them is small. Flip that with 1 - distance and you get a similarity score, where anything close to 1 means a strong match.
I found 0.7 to be a reliable threshold in my testing — chunks above that were almost always relevant. Anything below 0.5 and the retrieval was really stretching, pulling chunks that shared a keyword or two but weren't actually answering the question.
When that happens, the system prompt instruction ("if the context does not contain enough information, say so clearly") becomes important. A well-behaved model will tell the user it doesn't know rather than guess.
We also surface the source filename. Once you've ingested more than one document, users need to know whether that answer came from the architecture spec or the incident report.
*/
async function queryDocuments(question) {
// The question is embedded with the same model used at ingest time. This is
// not optional — vectors from two different models live in unrelated
// coordinate spaces, so comparing across them yields meaningless distances.
const questionEmbedding = await embedText(question);
// How this query works:
//
// embedding <=> $1 -> cosine DISTANCE (0 = identical, 2 = opposite)
// 1 - (that) -> cosine SIMILARITY, the human-friendly 0..1 score
//
// ORDER BY uses the raw distance, ascending, because "closest first" is
// literally smallest distance first. We order by the distance rather than by
// the derived `similarity` alias so the expression stays in the form a
// pgvector index can serve; ordering by 1 - distance inverts the direction
// and prevents the index from being used.
//
// Note there is no similarity threshold in SQL: we always take the best 5
// and let the model judge. The system prompt in generateAnswer() instructs it
// to admit when the context is insufficient, which handles weak matches more
// gracefully than returning nothing at all.
const { rows } = await pool.query(
`SELECT content, source,
1 - (embedding <=> $1::vector) AS similarity
FROM documents
ORDER BY embedding <=> $1::vector
LIMIT 5`,
[JSON.stringify(questionEmbedding)],
);
if (rows.length === 0) {
return { answer: "No relevant documents found.", sources: [] };
}
const context = rows.map((r) => r.content).join("\n\n---\n\n");
const answer = await generateAnswer(context, question);
return {
answer,
sources: [...new Set(rows.map((r) => r.source))],
topSimilarity: parseFloat(rows[0].similarity).toFixed(3),
};
}
module.exports = { queryDocuments };