- Citation hit — retrieved chunk IDs must include the gold chunk_id. This is the first gate.
- Faithfulness — every claim in the answer must be supportable by retrieved text. Unsupported claims fail the row.
- Answer completeness — required entities from gold ("14 days", "store credit") must appear.
{
"id": "refund-window",
"question": "What is the refund window?",
"gold_chunk_id": "policy-refunds-v3",
"must_include": ["14 days"],
"forbidden": ["30 days"]
}
type Row = {
gold_chunk_id: string;
must_include: string[];
forbidden: string[];
};
export function scoreRagRow(
row: Row,
retrievedIds: string[],
answer: string
) {
const citationHit = retrievedIds.includes(row.gold_chunk_id) ? 1 : 0;
const completeness = row.must_include.every((s) =>
answer.toLowerCase().includes(s.toLowerCase())
)
? 1
: 0;
const clean = row.forbidden.every(
(s) => !answer.toLowerCase().includes(s.toLowerCase())
)
? 1
: 0;
return { citationHit, completeness, clean };
}
export function gate(rows: ReturnType<typeof scoreRagRow>[]) {
const citation = rows.reduce((a, r) => a + r.citationHit, 0) / rows.length;
if (citation < 0.85) throw new Error(`citation hit ${citation} < 0.85`);
}
Run this on every index or prompt change. Do not wait for a monthly eval day. Not sure where your RAG quality stands? Start with the free QA maturity assessment. Productionise it through AI quality engineering services or a self-hosted control plane on the Enterprise page.