From aaf561846618ce16ac9da855e41a3d05f58a4cda Mon Sep 17 00:00:00 2001 From: Luiz Gustavo Date: Mon, 25 May 2026 04:05:33 -0300 Subject: [PATCH] fix(case-writer): doc-scoped retrieval for per-document case files MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A per-document case_report grounded itself via hybridSearch keyed on the document's auto-derived topic ("Fbi Photo B20", "Doc 59 214434 …"). That garbage topic has no semantic neighbours, so the dense gate rejected every chunk and the narrator saw zero scenes — skipping 61 of 75 batch documents that in fact had dozens of embedded chunks. When doc_id is set, fetch the document's own substantive chunks in reading order (fetchDocChunks) instead of searching the corpus. Corpus-wide topic reports still use hybridSearch. Verified post-deploy: fbi-photo-b19 0→10 scenes, doc-65 →24, dow-uap-d44 →11. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../src/detectives/case_writer.ts | 33 ++++++++++++------- investigator-runtime/src/lib/search.ts | 29 ++++++++++++++++ 2 files changed, 50 insertions(+), 12 deletions(-) diff --git a/investigator-runtime/src/detectives/case_writer.ts b/investigator-runtime/src/detectives/case_writer.ts index 04ae51d..3389c0a 100644 --- a/investigator-runtime/src/detectives/case_writer.ts +++ b/investigator-runtime/src/detectives/case_writer.ts @@ -16,7 +16,7 @@ import { audit } from "../lib/audit"; import { callClaude } from "../lib/claude"; import { env } from "../lib/env"; import { query } from "../lib/pg"; -import { hybridSearch, type SearchHit } from "../lib/search"; +import { fetchDocChunks, hybridSearch, type SearchHit } from "../lib/search"; import { writeCaseReport } from "../tools/write_case_report"; const HERE = path.dirname(fileURLToPath(import.meta.url)); @@ -286,18 +286,27 @@ export async function runCaseWriter(task: CaseWriterTask): Promise< const filter = `%${topic.toLowerCase()}%`; - // Grounding pass — retrieve top scenes from the corpus via hybrid_search. - // This is what gives the narrator real verbatim material to weave. Without - // this, the case-writer only sees pre-digested artefacts (which is what - // produced the academic prose in v1). - const scenes = await hybridSearch({ - query: topic, lang, - doc_id: task.doc_id ?? null, - top_k: 18, - recall_k: 80, - max_dense_dist: 0.55, - }).catch(() => [] as SearchHit[]); + // Grounding pass — assemble the scenes the narrator weaves from. + // + // For a per-document case file (doc_id set) we pull THIS document's own + // chunks in reading order. A hybridSearch keyed on the document's + // auto-derived topic ("Fbi Photo B20", "Doc 59 214434 …") returns zero + // hits even though the doc has dozens of embedded chunks — the dense gate + // rejects them all because the garbage topic has no semantic neighbours. + // That single bug skipped 61 of 75 batch documents on its own. + // + // For a corpus-wide topic report (no doc_id) the semantic search is + // exactly right — we want the strongest chunks across all documents. const docIdFilter = task.doc_id ?? null; + const scenes = docIdFilter + ? await fetchDocChunks(docIdFilter, lang, 24).catch(() => [] as SearchHit[]) + : await hybridSearch({ + query: topic, lang, + doc_id: null, + top_k: 18, + recall_k: 80, + max_dense_dist: 0.55, + }).catch(() => [] as SearchHit[]); // Pull artefacts SEQUENTIALLY. The investigator role has rolconnlimit=4 and // pool.max=4; Promise.all of 5 queries × max_parallel=2 jobs would demand diff --git a/investigator-runtime/src/lib/search.ts b/investigator-runtime/src/lib/search.ts index 8f15ae5..15e664c 100644 --- a/investigator-runtime/src/lib/search.ts +++ b/investigator-runtime/src/lib/search.ts @@ -67,6 +67,35 @@ export interface HybridSearchOpts { max_dense_dist?: number; } +/** + * Fetch a single document's own chunks in reading order — no semantic + * gating. For a per-document case file the narrator wants THIS document's + * substance, not a corpus search; a hybridSearch keyed on the document's + * (often garbage) auto-derived topic returns zero hits even though the + * doc has dozens of embedded chunks. We pull substantive `is_searchable` + * chunks ordered by `order_global` so the document tells its own story in + * sequence. + */ +export async function fetchDocChunks( + doc_id: string, + _lang: "pt" | "en" = "pt", + limit = 24, +): Promise { + if (!doc_id) return []; + return await query( + `SELECT chunk_pk, doc_id, chunk_id, page, type, bbox, + content_en, content_pt, classification, + 1.0::float8 AS score, NULL::int AS bm25_rank, NULL::int AS dense_rank + FROM public.chunks + WHERE doc_id = $1 + AND is_searchable = TRUE + AND length(COALESCE(content_en,'') || COALESCE(content_pt,'')) > 40 + ORDER BY order_global ASC NULLS LAST, page ASC, order_in_page ASC + LIMIT $2`, + [doc_id, limit], + ); +} + export async function hybridSearch(opts: HybridSearchOpts): Promise { const { query: q,