Compare commits

...

3 commits

Author SHA1 Message Date
Luiz Gustavo
6acc587dd5 fix(case-writer): rank doc chunks by richness, not just reading order
Some checks failed
CI / Web — typecheck + lint + build (push) Failing after 34s
CI / Scripts — Python smoke (push) Failing after 4s
CI / Web — npm audit (push) Failing after 37s
CI / Retrieval — golden set (Recall@5 + MRR) (push) Failing after 5s
fetchDocChunks took the first N chunks by order_global. On long files the
opening chunks are cover pages, routing slips, classification stamps and
redaction boxes — administrative front matter, not narrative. A 1001-chunk
FBI dossier (doc-65 section-7) handed the narrator 6 form_fields + 4
redaction_blocks + 2 letterheads and only 3 prose paragraphs, so it
refused with INSUFFICIENT_ARTEFACTS — burying first-person testimony
(e.g. the Flatwoods photographs) that sat deeper in the file.

Now select the richest substantive chunks (by content length, redaction
boxes last), then re-sort the winners into reading order for narrative
coherence. Validated: section-7 goes from 3 prose paragraphs to 22
(avg 1926 chars). 32 wrongly-refused rich documents re-queued.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-25 05:29:35 -03:00
Luiz Gustavo
e45f1c04d0 fix(seo): use body H1 as case-file <title> and OG title
generateMetadata still derived the page <title>/OG title from the generic
auto frontmatter topic ("Dow Uap D44 …"). Search engines, link cards, and
AI crawlers show that string, so it must be the human magazine headline the
narrator wrote as the body H1. Mirror the page component's H1 extraction.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-25 04:09:15 -03:00
Luiz Gustavo
aaf5618466 fix(case-writer): doc-scoped retrieval for per-document case files
A per-document case_report grounded itself via hybridSearch keyed on the
document's auto-derived topic ("Fbi Photo B20", "Doc 59 214434 …"). That
garbage topic has no semantic neighbours, so the dense gate rejected every
chunk and the narrator saw zero scenes — skipping 61 of 75 batch documents
that in fact had dozens of embedded chunks.

When doc_id is set, fetch the document's own substantive chunks in reading
order (fetchDocChunks) instead of searching the corpus. Corpus-wide topic
reports still use hybridSearch.

Verified post-deploy: fbi-photo-b19 0→10 scenes, doc-65 →24, dow-uap-d44 →11.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-25 04:05:33 -03:00
3 changed files with 73 additions and 13 deletions

View file

@ -16,7 +16,7 @@ import { audit } from "../lib/audit";
import { callClaude } from "../lib/claude";
import { env } from "../lib/env";
import { query } from "../lib/pg";
import { hybridSearch, type SearchHit } from "../lib/search";
import { fetchDocChunks, hybridSearch, type SearchHit } from "../lib/search";
import { writeCaseReport } from "../tools/write_case_report";
const HERE = path.dirname(fileURLToPath(import.meta.url));
@ -286,18 +286,27 @@ export async function runCaseWriter(task: CaseWriterTask): Promise<
const filter = `%${topic.toLowerCase()}%`;
// Grounding pass — retrieve top scenes from the corpus via hybrid_search.
// This is what gives the narrator real verbatim material to weave. Without
// this, the case-writer only sees pre-digested artefacts (which is what
// produced the academic prose in v1).
const scenes = await hybridSearch({
query: topic, lang,
doc_id: task.doc_id ?? null,
top_k: 18,
recall_k: 80,
max_dense_dist: 0.55,
}).catch(() => [] as SearchHit[]);
// Grounding pass — assemble the scenes the narrator weaves from.
//
// For a per-document case file (doc_id set) we pull THIS document's own
// chunks in reading order. A hybridSearch keyed on the document's
// auto-derived topic ("Fbi Photo B20", "Doc 59 214434 …") returns zero
// hits even though the doc has dozens of embedded chunks — the dense gate
// rejects them all because the garbage topic has no semantic neighbours.
// That single bug skipped 61 of 75 batch documents on its own.
//
// For a corpus-wide topic report (no doc_id) the semantic search is
// exactly right — we want the strongest chunks across all documents.
const docIdFilter = task.doc_id ?? null;
const scenes = docIdFilter
? await fetchDocChunks(docIdFilter, lang, 24).catch(() => [] as SearchHit[])
: await hybridSearch({
query: topic, lang,
doc_id: null,
top_k: 18,
recall_k: 80,
max_dense_dist: 0.55,
}).catch(() => [] as SearchHit[]);
// Pull artefacts SEQUENTIALLY. The investigator role has rolconnlimit=4 and
// pool.max=4; Promise.all of 5 queries × max_parallel=2 jobs would demand

View file

@ -67,6 +67,50 @@ export interface HybridSearchOpts {
max_dense_dist?: number;
}
/**
* Fetch a single document's own chunks no semantic gating. For a
* per-document case file the narrator wants THIS document's substance, not
* a corpus search; a hybridSearch keyed on the document's (often garbage)
* auto-derived topic returns zero hits even though the doc has dozens of
* embedded chunks.
*
* We pick the most substantive chunks (by content length, deprioritising
* pure redaction boxes) and THEN present them in reading order. Naively
* taking the first N by `order_global` starves the narrator on long files:
* the opening chunks of a 1000-chunk FBI dossier are cover pages, routing
* slips, classification stamps and redaction boxes administrative front
* matter, not narrative. The substance sits deeper in the file, so a
* length-ranked pick surfaces it regardless of position, while the final
* reading-order sort keeps the story coherent.
*/
export async function fetchDocChunks(
doc_id: string,
_lang: "pt" | "en" = "pt",
limit = 24,
): Promise<SearchHit[]> {
if (!doc_id) return [];
return await query<SearchHit>(
`WITH ranked AS (
SELECT chunk_pk, doc_id, chunk_id, page, type, bbox,
content_en, content_pt, classification,
order_global, order_in_page,
length(COALESCE(content_en,'') || COALESCE(content_pt,'')) AS richness
FROM public.chunks
WHERE doc_id = $1
AND is_searchable = TRUE
AND length(COALESCE(content_en,'') || COALESCE(content_pt,'')) > 40
ORDER BY (type = 'redaction') ASC, richness DESC
LIMIT $2
)
SELECT chunk_pk, doc_id, chunk_id, page, type, bbox,
content_en, content_pt, classification,
1.0::float8 AS score, NULL::int AS bm25_rank, NULL::int AS dense_rank
FROM ranked
ORDER BY order_global ASC NULLS LAST, page ASC, order_in_page ASC`,
[doc_id, limit],
);
}
export async function hybridSearch(opts: HybridSearchOpts): Promise<SearchHit[]> {
const {
query: q,

View file

@ -81,7 +81,14 @@ export async function generateMetadata(
const c = await loadCase(slug);
if (!c) return { title: "Case file not found" };
const title = locale === "pt-br" ? (c.fm.topic_pt_br ?? c.fm.topic ?? slug) : (c.fm.topic ?? slug);
// Prefer the narrator's body H1 (the magazine headline) over the generic
// auto-derived frontmatter topic ("Dow Uap D44 …"). This is what search
// engines and link cards show, so it must be the human title. Two H1s when
// present: EN then PT-BR; fall back across them, then to frontmatter.
const h1s = [...c.body.matchAll(/^#\s+(.+)$/gm)].map((m) => m[1].trim());
const title = locale === "pt-br"
? (h1s[1] ?? h1s[0] ?? c.fm.topic_pt_br ?? c.fm.topic ?? slug)
: (h1s[0] ?? c.fm.topic ?? slug);
const desc = pickLead(c.body, locale).slice(0, 200);
const canonical = `${SITE_URL}/c/${slug}`;
// OG image — use the case's editorial illustration when present. WhatsApp,