An AI feature that got worse and nobody noticed, because the only test was somebody trying it once.
Add "Model output evaluation tests" from ntent to this repo.
Fetch https://ntent.app/r/f/eval-harness as plain text. Write it verbatim to evals/eval.test.mjs.
It needs an AI feature that reaches users and evals/judge.mjs and evals/report.mjs and evals/types.mjs. If this repo does not have them, stop and tell me.
Then check it: node --test evals/eval.test.mjs passes. That is the harness proving ITSELF, on stubbed judge responses: it is what makes the scoring trustworthy, not a measurement of your feature. Running your own questions is the integration step, and it is yours to write: give judge() a judgeFn that calls your provider, feed it real responses with their retrieved sources, and put the summary somewhere you keep.Reads https://ntent.app/r/f/eval-harness
Read the code
eval.test.mjsevals/eval.test.mjs
278 lines
/**
* Tests for the eval harness itself.
*
* Yes, tests for the tests. The scoring logic decides whether your build is red,
* and a bug here is worse than no harness at all: it either blocks good releases
* or silently passes bad ones. Two of the assertions below are for bugs that are
* easy to write and hard to notice.
*
* Run: node --test evals/eval.test.mjs
*/
import { test } from 'node:test'
import assert from 'node:assert/strict'
import { isPassing, isCriticalFailure, summarise } from './report.mjs'
import { mechanicalChecks, judge, buildUserPrompt } from './judge.mjs'
import { GATING_DIMENSIONS, MIN_DIMENSION_SCORE } from './types.mjs'
/** A complete positive judge reply, which is now the bar for a usable result. */
const fullScores = { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 5 }
const fullReasoning = Object.fromEntries(Object.keys(fullScores).map((d) => [d, 'fine']))
const fullReply = JSON.stringify({ scores: fullScores, reasoning: fullReasoning })
/** A response carrying what retrieval actually returned, text included. */
const grounded = (over = {}) => ({
answer: 'You have 30 days.',
confidence: 'high',
citations: [{ title: 'Refunds', source: 'policy/refunds.md' }],
retrieved: [{ id: 'S1', title: 'Refunds', text: 'Annual plans may be refunded within 30 days.' }],
...over,
})
const positive = (over = {}) => ({
id: 'q1',
query: 'refund window?',
category: 'positive',
segment: 'billing',
scores: { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 5 },
...over,
})
// ------------------------------------------------------------- gating -------
test('a low score on a gating dimension fails the question', () => {
assert.equal(isPassing(positive({ scores: { grounding: 2, relevance: 5, calibration: 5, citations: 5 } })), false)
})
test('a low score on a NON-gating dimension does not fail the question', () => {
// The whole point of the gating/reported split. If this ever starts failing,
// someone has added `tone` to GATING_DIMENSIONS and the suite is about to
// become flaky.
assert.equal(isPassing(positive({ scores: { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 1 } })), true)
})
test('a negative case scored on only two dimensions still passes', () => {
// The easy bug: treating an unscored dimension as zero. That fails every
// negative case in the suite, and since negative cases are the ones that
// catch fabrication, you would then delete them to get green.
const negative = {
id: 'q2',
category: 'negative',
segment: 'billing',
query: 'what is our stance on quantum computing?',
scores: { grounding: 5, calibration: 5 },
}
assert.equal(isPassing(negative), true)
})
test('a judge result with no scores in it FAILS, rather than passing on silence', () => {
// The one that made the harness worthless: skipping an absent dimension is
// correct for a negative case and became a general licence, so {"scores":{}}
// passed. An empty result is most often a broken judge, which meant a broken
// judge reported a perfect suite.
assert.equal(isPassing(positive({ scores: {} })), false)
assert.equal(isPassing(positive({ scores: undefined })), false)
assert.equal(
isPassing({ id: 'n', category: 'negative', segment: 'x', scores: {} }),
false,
'a negative case owes two dimensions, and owing two is not owing none'
)
})
test('a score that is not a number in range is not a score', () => {
assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: '5' } })), false)
assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: NaN } })), false)
assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: 9 } })), false)
})
test('an unassessable dimension fails the question rather than being skipped', () => {
// "We could not check" and "we checked and it was fine" are different
// sentences and only one of them is a pass.
assert.equal(isPassing(positive({ unassessable: ['grounding', 'citations'] })), false)
})
test('a mechanical failure fails the question regardless of scores', () => {
assert.equal(isPassing(positive({ mechanicalFailures: ['contains forbidden claim: "90 days"'] })), false)
})
test('the floor is inclusive', () => {
const atFloor = Object.fromEntries(GATING_DIMENSIONS.map((d) => [d, MIN_DIMENSION_SCORE]))
assert.equal(isPassing(positive({ scores: atFloor })), true)
const belowFloor = { ...atFloor, grounding: MIN_DIMENSION_SCORE - 1 }
assert.equal(isPassing(positive({ scores: belowFloor })), false)
})
// ---------------------------------------------------- mechanical checks -----
test('mustMention and mustNotMention are enforced without a model', () => {
const q = { rubric: { shouldAnswer: true, mustMention: ['30 days'], mustNotMention: ['90 days'] } }
assert.deepEqual(mechanicalChecks(q, { answer: 'You have 30 days to request a refund.' }), [])
assert.equal(mechanicalChecks(q, { answer: 'You have 90 days.' }).length, 2)
})
test('answering an out-of-corpus question with confidence is a mechanical failure', () => {
const q = { rubric: { shouldAnswer: false } }
const bad = mechanicalChecks(q, { answer: 'Our policy is 45 days.', confidence: 'high' })
assert.equal(bad.length, 1)
const good = mechanicalChecks(q, { answer: null, confidence: 'none' })
assert.equal(good.length, 0)
})
// ------------------------------------------------------------- summary ------
test('the summary breaks down by segment, worst first in the report', () => {
const s = summarise([
positive({ id: 'a', segment: 'billing' }),
positive({ id: 'b', segment: 'billing', scores: { grounding: 1, relevance: 5, calibration: 5, citations: 5 } }),
positive({ id: 'c', segment: 'onboarding' }),
])
assert.equal(s.total, 3)
assert.equal(s.passed, 2)
assert.equal(s.bySegment.billing.passRate, 0.5)
assert.equal(s.bySegment.onboarding.passRate, 1)
})
test('averages skip dimensions a question was not scored on', () => {
const s = summarise([
positive({ scores: { grounding: 4, relevance: 4, calibration: 4, citations: 4, tone: 2 } }),
{ id: 'n', category: 'negative', segment: 'x', scores: { grounding: 5, calibration: 5 } },
])
// grounding averages over both, tone over only the one that had it.
assert.equal(s.avgScores.grounding, 4.5)
assert.equal(s.avgScores.tone, 2)
})
test('the suite verdict uses the pass rate, not the average score', () => {
// A suite can have good averages and still be failing, if the failures are
// concentrated. Averaging hides that; the pass rate does not.
const results = Array.from({ length: 10 }, (_, i) =>
positive({ id: `q${i}`, scores: i < 8 ? { grounding: 5, relevance: 5, calibration: 5, citations: 5 } : { grounding: 1, relevance: 1, calibration: 1, citations: 1 } })
)
const s = summarise(results)
assert.equal(s.passRate, 0.8)
assert.equal(s.ok, false, '80% is below the 90% floor even though 8 of 10 are perfect')
})
test('one critical failure fails the suite whatever the pass rate says', () => {
// Nineteen clean answers and one that leaked a forbidden claim reported 95%
// and PASS. The floor is for judge noise, and a deterministic failure on a
// case somebody marked critical is not noise.
const results = Array.from({ length: 19 }, (_, i) => positive({ id: `ok${i}` }))
results.push(positive({ id: 'leak', critical: true, mechanicalFailures: ['contains forbidden claim: "90 days"'] }))
const s = summarise(results)
assert.equal(s.passRate, 0.95, 'the rate is still reported honestly')
assert.deepEqual(s.criticalFailures, ['leak'])
assert.equal(s.ok, false)
})
test('a mechanical failure is critical by default, because it is deterministic', () => {
assert.equal(isCriticalFailure(positive({ mechanicalFailures: ['x'] })), true)
assert.equal(isCriticalFailure(positive({ critical: true, scores: { ...fullScores, grounding: 1 } })), true, 'a critical case fails critically for any reason')
assert.equal(isCriticalFailure(positive({ scores: { ...fullScores, grounding: 1 } })), false, 'an ordinary low score is what the floor is for')
assert.equal(isCriticalFailure(positive({ critical: true })), false, 'a passing critical case is just a pass')
})
test('a result the harness could not measure fails the suite, separately from a low score', () => {
const results = Array.from({ length: 19 }, (_, i) => positive({ id: `ok${i}` }))
results.push(positive({ id: 'blind', unassessable: ['grounding', 'citations'] }))
const s = summarise(results)
assert.deepEqual(s.invalid, ['blind'])
assert.deepEqual(s.criticalFailures, [])
assert.equal(s.ok, false, '"we did not check" is not a data point in a pass rate')
})
test('no results is not a pass', () => {
assert.equal(summarise([]).ok, false)
})
test('the summary carries what produced it', () => {
const s = summarise([positive()], { model: 'claude-fable-5-1', prompt: 'v12', corpus: '2026-09-01', judge: 'claude-sonnet-5', commit: 'abc123' })
assert.equal(s.meta.prompt, 'v12')
assert.equal(s.meta.commit, 'abc123')
})
// --------------------------------------------------------------- judge ------
test('the judge tolerates a fenced JSON response', async () => {
const stub = async () => '```json\n' + fullReply + '\n```'
const out = await judge({ query: 'x', rubric: { shouldAnswer: true } }, grounded(), stub)
assert.equal(out.scores.grounding, 5)
})
test('the judge retries once before giving up', async () => {
let calls = 0
const flaky = async () => {
calls++
if (calls === 1) throw new Error('rate limited')
return fullReply
}
const out = await judge({ query: 'x', rubric: { shouldAnswer: true } }, grounded(), flaky)
assert.equal(calls, 2)
assert.equal(out.scores.grounding, 5)
})
test('a negative question gets the negative judge prompt', async () => {
let systemSeen = ''
const spy = async ({ system }) => {
systemSeen = system
return '{"scores":{"grounding":5,"calibration":5},"reasoning":{"grounding":"declined","calibration":"none"}}'
}
await judge({ query: 'x', rubric: { shouldAnswer: false } }, { answer: null }, spy)
assert.match(systemSeen, /does NOT cover/, 'a negative case must not be scored on relevance or tone')
})
test('the judge refuses a reply that skipped a dimension it was asked for', async () => {
const thin = async () => '{"scores":{"grounding":5},"reasoning":{"grounding":"fine"}}'
await assert.rejects(
judge({ id: 'q1', query: 'x', category: 'positive', rubric: { shouldAnswer: true } }, grounded(), thin),
/unusable/,
'a partial result is a harness failure, and it has to read as one'
)
})
test('the judge refuses an empty result, which is what a failing judge returns', async () => {
const empty = async () => '{"scores":{},"reasoning":{}}'
await assert.rejects(
judge({ id: 'q1', query: 'x', rubric: { shouldAnswer: true } }, grounded(), empty),
/unusable/
)
})
test('a score with no reasoning behind it is not accepted', async () => {
const mute = async () => JSON.stringify({ scores: fullScores, reasoning: {} })
await assert.rejects(
judge({ id: 'q1', query: 'x', rubric: { shouldAnswer: true } }, grounded(), mute),
/no reasoning/,
'an unexplained score cannot be read and dismissed, which is what makes judging tolerable'
)
})
// ------------------------------------------------------------- evidence -----
test('the judge is shown the retrieved passages, not just the citation titles', () => {
// The failure this locks down: the prompt asked whether every assertion was
// supported by the sources and then sent only their names. Any grounding
// score that came back was about how plausible a filename looked.
const prompt = buildUserPrompt({ query: 'refund window?', rubric: { shouldAnswer: true } }, grounded())
assert.match(prompt, /Annual plans may be refunded within 30 days/, 'the source TEXT has to be in there')
assert.match(prompt, /\[S1\]/, 'with a stable id, so a reasoning line can name the passage')
})
test('with no retrieved sources, grounding and citations are reported unassessable', async () => {
let userSeen = ''
const spy = async ({ user }) => {
userSeen = user
return JSON.stringify({
scores: { relevance: 5, calibration: 5, tone: 5 },
reasoning: { relevance: 'ok', calibration: 'ok', tone: 'ok' },
})
}
const out = await judge(
{ id: 'q1', query: 'x', rubric: { shouldAnswer: true } },
{ answer: 'y', citations: [{ title: 'Something', source: 'a.md' }] },
spy
)
assert.match(userSeen, /none recorded/, 'and the judge is told not to guess at them')
assert.deepEqual(out.unassessable, ['grounding', 'citations'])
assert.equal(isPassing({ category: 'positive', ...out }), false, 'unassessable is not a pass')
})
Success check: node --test evals/eval.test.mjs passes. That is the harness proving ITSELF, on stubbed judge responses: it is what makes the scoring trustworthy, not a measurement of your feature. Running your own questions is the integration step, and it is yours to write: give judge() a judgeFn that calls your provider, feed it real responses with their retrieved sources, and put the summary somewhere you keep.
Included in
- Any tier plan with ?with=ai, because a model’s output reaches a user