eval.test.mjs

Model output evaluation tests

+ Evalsengineering

An AI feature that got worse and nobody noticed, because the only test was somebody trying it once.

Add to your project
Add "Model output evaluation tests" from ntent to this repo.

Fetch https://ntent.app/r/f/eval-harness as plain text. Write it verbatim to evals/eval.test.mjs.

It needs an AI feature that reaches users and evals/judge.mjs and evals/report.mjs and evals/types.mjs. If this repo does not have them, stop and tell me.

Then check it: node --test evals/eval.test.mjs passes. That is the harness proving ITSELF, on stubbed judge responses: it is what makes the scoring trustworthy, not a measurement of your feature. Running your own questions is the integration step, and it is yours to write: give judge() a judgeFn that calls your provider, feed it real responses with their retrieved sources, and put the summary somewhere you keep.

Reads https://ntent.app/r/f/eval-harness

Read the code
eval.test.mjsevals/eval.test.mjs
278 lines
/**
 * Tests for the eval harness itself.
 *
 * Yes, tests for the tests. The scoring logic decides whether your build is red,
 * and a bug here is worse than no harness at all: it either blocks good releases
 * or silently passes bad ones. Two of the assertions below are for bugs that are
 * easy to write and hard to notice.
 *
 * Run:  node --test evals/eval.test.mjs
 */
import { test } from 'node:test'
import assert from 'node:assert/strict'
import { isPassing, isCriticalFailure, summarise } from './report.mjs'
import { mechanicalChecks, judge, buildUserPrompt } from './judge.mjs'
import { GATING_DIMENSIONS, MIN_DIMENSION_SCORE } from './types.mjs'

/** A complete positive judge reply, which is now the bar for a usable result. */
const fullScores = { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 5 }
const fullReasoning = Object.fromEntries(Object.keys(fullScores).map((d) => [d, 'fine']))
const fullReply = JSON.stringify({ scores: fullScores, reasoning: fullReasoning })

/** A response carrying what retrieval actually returned, text included. */
const grounded = (over = {}) => ({
  answer: 'You have 30 days.',
  confidence: 'high',
  citations: [{ title: 'Refunds', source: 'policy/refunds.md' }],
  retrieved: [{ id: 'S1', title: 'Refunds', text: 'Annual plans may be refunded within 30 days.' }],
  ...over,
})

const positive = (over = {}) => ({
  id: 'q1',
  query: 'refund window?',
  category: 'positive',
  segment: 'billing',
  scores: { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 5 },
  ...over,
})

// ------------------------------------------------------------- gating -------

test('a low score on a gating dimension fails the question', () => {
  assert.equal(isPassing(positive({ scores: { grounding: 2, relevance: 5, calibration: 5, citations: 5 } })), false)
})

test('a low score on a NON-gating dimension does not fail the question', () => {
  // The whole point of the gating/reported split. If this ever starts failing,
  // someone has added `tone` to GATING_DIMENSIONS and the suite is about to
  // become flaky.
  assert.equal(isPassing(positive({ scores: { grounding: 5, relevance: 5, calibration: 5, citations: 5, tone: 1 } })), true)
})

test('a negative case scored on only two dimensions still passes', () => {
  // The easy bug: treating an unscored dimension as zero. That fails every
  // negative case in the suite, and since negative cases are the ones that
  // catch fabrication, you would then delete them to get green.
  const negative = {
    id: 'q2',
    category: 'negative',
    segment: 'billing',
    query: 'what is our stance on quantum computing?',
    scores: { grounding: 5, calibration: 5 },
  }
  assert.equal(isPassing(negative), true)
})

test('a judge result with no scores in it FAILS, rather than passing on silence', () => {
  // The one that made the harness worthless: skipping an absent dimension is
  // correct for a negative case and became a general licence, so {"scores":{}}
  // passed. An empty result is most often a broken judge, which meant a broken
  // judge reported a perfect suite.
  assert.equal(isPassing(positive({ scores: {} })), false)
  assert.equal(isPassing(positive({ scores: undefined })), false)
  assert.equal(
    isPassing({ id: 'n', category: 'negative', segment: 'x', scores: {} }),
    false,
    'a negative case owes two dimensions, and owing two is not owing none'
  )
})

test('a score that is not a number in range is not a score', () => {
  assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: '5' } })), false)
  assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: NaN } })), false)
  assert.equal(isPassing(positive({ scores: { ...fullScores, grounding: 9 } })), false)
})

test('an unassessable dimension fails the question rather than being skipped', () => {
  // "We could not check" and "we checked and it was fine" are different
  // sentences and only one of them is a pass.
  assert.equal(isPassing(positive({ unassessable: ['grounding', 'citations'] })), false)
})

test('a mechanical failure fails the question regardless of scores', () => {
  assert.equal(isPassing(positive({ mechanicalFailures: ['contains forbidden claim: "90 days"'] })), false)
})

test('the floor is inclusive', () => {
  const atFloor = Object.fromEntries(GATING_DIMENSIONS.map((d) => [d, MIN_DIMENSION_SCORE]))
  assert.equal(isPassing(positive({ scores: atFloor })), true)
  const belowFloor = { ...atFloor, grounding: MIN_DIMENSION_SCORE - 1 }
  assert.equal(isPassing(positive({ scores: belowFloor })), false)
})

// ---------------------------------------------------- mechanical checks -----

test('mustMention and mustNotMention are enforced without a model', () => {
  const q = { rubric: { shouldAnswer: true, mustMention: ['30 days'], mustNotMention: ['90 days'] } }
  assert.deepEqual(mechanicalChecks(q, { answer: 'You have 30 days to request a refund.' }), [])
  assert.equal(mechanicalChecks(q, { answer: 'You have 90 days.' }).length, 2)
})

test('answering an out-of-corpus question with confidence is a mechanical failure', () => {
  const q = { rubric: { shouldAnswer: false } }
  const bad = mechanicalChecks(q, { answer: 'Our policy is 45 days.', confidence: 'high' })
  assert.equal(bad.length, 1)
  const good = mechanicalChecks(q, { answer: null, confidence: 'none' })
  assert.equal(good.length, 0)
})

// ------------------------------------------------------------- summary ------

test('the summary breaks down by segment, worst first in the report', () => {
  const s = summarise([
    positive({ id: 'a', segment: 'billing' }),
    positive({ id: 'b', segment: 'billing', scores: { grounding: 1, relevance: 5, calibration: 5, citations: 5 } }),
    positive({ id: 'c', segment: 'onboarding' }),
  ])
  assert.equal(s.total, 3)
  assert.equal(s.passed, 2)
  assert.equal(s.bySegment.billing.passRate, 0.5)
  assert.equal(s.bySegment.onboarding.passRate, 1)
})

test('averages skip dimensions a question was not scored on', () => {
  const s = summarise([
    positive({ scores: { grounding: 4, relevance: 4, calibration: 4, citations: 4, tone: 2 } }),
    { id: 'n', category: 'negative', segment: 'x', scores: { grounding: 5, calibration: 5 } },
  ])
  // grounding averages over both, tone over only the one that had it.
  assert.equal(s.avgScores.grounding, 4.5)
  assert.equal(s.avgScores.tone, 2)
})

test('the suite verdict uses the pass rate, not the average score', () => {
  // A suite can have good averages and still be failing, if the failures are
  // concentrated. Averaging hides that; the pass rate does not.
  const results = Array.from({ length: 10 }, (_, i) =>
    positive({ id: `q${i}`, scores: i < 8 ? { grounding: 5, relevance: 5, calibration: 5, citations: 5 } : { grounding: 1, relevance: 1, calibration: 1, citations: 1 } })
  )
  const s = summarise(results)
  assert.equal(s.passRate, 0.8)
  assert.equal(s.ok, false, '80% is below the 90% floor even though 8 of 10 are perfect')
})

test('one critical failure fails the suite whatever the pass rate says', () => {
  // Nineteen clean answers and one that leaked a forbidden claim reported 95%
  // and PASS. The floor is for judge noise, and a deterministic failure on a
  // case somebody marked critical is not noise.
  const results = Array.from({ length: 19 }, (_, i) => positive({ id: `ok${i}` }))
  results.push(positive({ id: 'leak', critical: true, mechanicalFailures: ['contains forbidden claim: "90 days"'] }))
  const s = summarise(results)
  assert.equal(s.passRate, 0.95, 'the rate is still reported honestly')
  assert.deepEqual(s.criticalFailures, ['leak'])
  assert.equal(s.ok, false)
})

test('a mechanical failure is critical by default, because it is deterministic', () => {
  assert.equal(isCriticalFailure(positive({ mechanicalFailures: ['x'] })), true)
  assert.equal(isCriticalFailure(positive({ critical: true, scores: { ...fullScores, grounding: 1 } })), true, 'a critical case fails critically for any reason')
  assert.equal(isCriticalFailure(positive({ scores: { ...fullScores, grounding: 1 } })), false, 'an ordinary low score is what the floor is for')
  assert.equal(isCriticalFailure(positive({ critical: true })), false, 'a passing critical case is just a pass')
})

test('a result the harness could not measure fails the suite, separately from a low score', () => {
  const results = Array.from({ length: 19 }, (_, i) => positive({ id: `ok${i}` }))
  results.push(positive({ id: 'blind', unassessable: ['grounding', 'citations'] }))
  const s = summarise(results)
  assert.deepEqual(s.invalid, ['blind'])
  assert.deepEqual(s.criticalFailures, [])
  assert.equal(s.ok, false, '"we did not check" is not a data point in a pass rate')
})

test('no results is not a pass', () => {
  assert.equal(summarise([]).ok, false)
})

test('the summary carries what produced it', () => {
  const s = summarise([positive()], { model: 'claude-fable-5-1', prompt: 'v12', corpus: '2026-09-01', judge: 'claude-sonnet-5', commit: 'abc123' })
  assert.equal(s.meta.prompt, 'v12')
  assert.equal(s.meta.commit, 'abc123')
})

// --------------------------------------------------------------- judge ------

test('the judge tolerates a fenced JSON response', async () => {
  const stub = async () => '```json\n' + fullReply + '\n```'
  const out = await judge({ query: 'x', rubric: { shouldAnswer: true } }, grounded(), stub)
  assert.equal(out.scores.grounding, 5)
})

test('the judge retries once before giving up', async () => {
  let calls = 0
  const flaky = async () => {
    calls++
    if (calls === 1) throw new Error('rate limited')
    return fullReply
  }
  const out = await judge({ query: 'x', rubric: { shouldAnswer: true } }, grounded(), flaky)
  assert.equal(calls, 2)
  assert.equal(out.scores.grounding, 5)
})

test('a negative question gets the negative judge prompt', async () => {
  let systemSeen = ''
  const spy = async ({ system }) => {
    systemSeen = system
    return '{"scores":{"grounding":5,"calibration":5},"reasoning":{"grounding":"declined","calibration":"none"}}'
  }
  await judge({ query: 'x', rubric: { shouldAnswer: false } }, { answer: null }, spy)
  assert.match(systemSeen, /does NOT cover/, 'a negative case must not be scored on relevance or tone')
})

test('the judge refuses a reply that skipped a dimension it was asked for', async () => {
  const thin = async () => '{"scores":{"grounding":5},"reasoning":{"grounding":"fine"}}'
  await assert.rejects(
    judge({ id: 'q1', query: 'x', category: 'positive', rubric: { shouldAnswer: true } }, grounded(), thin),
    /unusable/,
    'a partial result is a harness failure, and it has to read as one'
  )
})

test('the judge refuses an empty result, which is what a failing judge returns', async () => {
  const empty = async () => '{"scores":{},"reasoning":{}}'
  await assert.rejects(
    judge({ id: 'q1', query: 'x', rubric: { shouldAnswer: true } }, grounded(), empty),
    /unusable/
  )
})

test('a score with no reasoning behind it is not accepted', async () => {
  const mute = async () => JSON.stringify({ scores: fullScores, reasoning: {} })
  await assert.rejects(
    judge({ id: 'q1', query: 'x', rubric: { shouldAnswer: true } }, grounded(), mute),
    /no reasoning/,
    'an unexplained score cannot be read and dismissed, which is what makes judging tolerable'
  )
})

// ------------------------------------------------------------- evidence -----

test('the judge is shown the retrieved passages, not just the citation titles', () => {
  // The failure this locks down: the prompt asked whether every assertion was
  // supported by the sources and then sent only their names. Any grounding
  // score that came back was about how plausible a filename looked.
  const prompt = buildUserPrompt({ query: 'refund window?', rubric: { shouldAnswer: true } }, grounded())
  assert.match(prompt, /Annual plans may be refunded within 30 days/, 'the source TEXT has to be in there')
  assert.match(prompt, /\[S1\]/, 'with a stable id, so a reasoning line can name the passage')
})

test('with no retrieved sources, grounding and citations are reported unassessable', async () => {
  let userSeen = ''
  const spy = async ({ user }) => {
    userSeen = user
    return JSON.stringify({
      scores: { relevance: 5, calibration: 5, tone: 5 },
      reasoning: { relevance: 'ok', calibration: 'ok', tone: 'ok' },
    })
  }
  const out = await judge(
    { id: 'q1', query: 'x', rubric: { shouldAnswer: true } },
    { answer: 'y', citations: [{ title: 'Something', source: 'a.md' }] },
    spy
  )
  assert.match(userSeen, /none recorded/, 'and the judge is told not to guess at them')
  assert.deepEqual(out.unassessable, ['grounding', 'citations'])
  assert.equal(isPassing({ category: 'positive', ...out }), false, 'unassessable is not a pass')
})

Success check: node --test evals/eval.test.mjs passes. That is the harness proving ITSELF, on stubbed judge responses: it is what makes the scoring trustworthy, not a measurement of your feature. Running your own questions is the integration step, and it is yours to write: give judge() a judgeFn that calls your provider, feed it real responses with their retrieved sources, and put the summary somewhere you keep.

Included in