diff --git a/investigator-runtime/src/lib/search.ts b/investigator-runtime/src/lib/search.ts index 15e664c..df8b7f8 100644 --- a/investigator-runtime/src/lib/search.ts +++ b/investigator-runtime/src/lib/search.ts @@ -68,13 +68,20 @@ export interface HybridSearchOpts { } /** - * Fetch a single document's own chunks in reading order — no semantic - * gating. For a per-document case file the narrator wants THIS document's - * substance, not a corpus search; a hybridSearch keyed on the document's - * (often garbage) auto-derived topic returns zero hits even though the - * doc has dozens of embedded chunks. We pull substantive `is_searchable` - * chunks ordered by `order_global` so the document tells its own story in - * sequence. + * Fetch a single document's own chunks — no semantic gating. For a + * per-document case file the narrator wants THIS document's substance, not + * a corpus search; a hybridSearch keyed on the document's (often garbage) + * auto-derived topic returns zero hits even though the doc has dozens of + * embedded chunks. + * + * We pick the most substantive chunks (by content length, deprioritising + * pure redaction boxes) and THEN present them in reading order. Naively + * taking the first N by `order_global` starves the narrator on long files: + * the opening chunks of a 1000-chunk FBI dossier are cover pages, routing + * slips, classification stamps and redaction boxes — administrative front + * matter, not narrative. The substance sits deeper in the file, so a + * length-ranked pick surfaces it regardless of position, while the final + * reading-order sort keeps the story coherent. */ export async function fetchDocChunks( doc_id: string, @@ -83,15 +90,23 @@ export async function fetchDocChunks( ): Promise { if (!doc_id) return []; return await query( - `SELECT chunk_pk, doc_id, chunk_id, page, type, bbox, + `WITH ranked AS ( + SELECT chunk_pk, doc_id, chunk_id, page, type, bbox, + content_en, content_pt, classification, + order_global, order_in_page, + length(COALESCE(content_en,'') || COALESCE(content_pt,'')) AS richness + FROM public.chunks + WHERE doc_id = $1 + AND is_searchable = TRUE + AND length(COALESCE(content_en,'') || COALESCE(content_pt,'')) > 40 + ORDER BY (type = 'redaction') ASC, richness DESC + LIMIT $2 + ) + SELECT chunk_pk, doc_id, chunk_id, page, type, bbox, content_en, content_pt, classification, 1.0::float8 AS score, NULL::int AS bm25_rank, NULL::int AS dense_rank - FROM public.chunks - WHERE doc_id = $1 - AND is_searchable = TRUE - AND length(COALESCE(content_en,'') || COALESCE(content_pt,'')) > 40 - ORDER BY order_global ASC NULLS LAST, page ASC, order_in_page ASC - LIMIT $2`, + FROM ranked + ORDER BY order_global ASC NULLS LAST, page ASC, order_in_page ASC`, [doc_id, limit], ); }