report.mjs

Evaluation scoring and report

+ Evalsengineering

A single pass rate that says something broke but not what. A grading result with no scores in it counting as a pass. One answer in 20 saying something it must never say, while the report still shows 95% and PASS.

Add to your project
Add "Evaluation scoring and report" from ntent to this repo.

Fetch https://ntent.app/r/f/eval-report as plain text. Write it verbatim to evals/report.mjs.

It needs evals/types.mjs. If this repo does not have it, stop and tell me.

Then check it: The report breaks down by segment and by category. Feed it a result whose gating dimensions are missing and the question fails rather than passes. Feed it nineteen passes and one case marked critical that failed, and the suite fails at 95%.

Reads https://ntent.app/r/f/eval-report

Read the code
report.mjsevals/report.mjs
194 lines
/**
 * Turning scores into a verdict, and into something someone can act on.
 *
 * The rule this file exists for: an aggregate pass rate tells you something
 * broke. A per-segment breakdown tells you WHAT broke, which is the difference
 * between a red build you can fix and a red build you re-run hoping it passes.
 */
import {
  GATING_DIMENSIONS,
  MECHANICAL_FAILURES_ARE_CRITICAL,
  MIN_DIMENSION_SCORE,
  PASS_RATE,
  REQUIRED_DIMENSIONS,
  SCORE_RANGE,
} from './types.mjs'

/**
 * The gating dimensions this particular question owes a score on.
 *
 * A negative case is scored on two dimensions by design, so the other three are
 * not missing. Everything a question's category DOES list has to be there.
 */
export function requiredGating(result) {
  const category = result.category ?? 'positive'
  const required = REQUIRED_DIMENSIONS[category] ?? REQUIRED_DIMENSIONS.positive
  return GATING_DIMENSIONS.filter((d) => required.includes(d))
}

/**
 * A question passes when every gating dimension its category requires was
 * scored, that score is a real number in range and meets the floor, and no
 * mechanical check failed.
 *
 * THE BUG THIS REPLACES. The old version skipped `scores[d] === undefined`,
 * which is right for a negative case and catastrophic as a general rule: a
 * judge returning {"scores":{}} passed, and so did one returning nothing at
 * all. The most likely cause of an empty result is the judge failing, so the
 * suite reported a clean sweep exactly when it had measured nothing. "We did
 * not check" must never render as "we checked and it was fine".
 *
 * `unassessable` is the same principle said out loud: grounding cannot be
 * scored when no retrieved sources were recorded, so the question fails with a
 * reason rather than passing on a dimension nobody could look at.
 */
export function isPassing(result) {
  if (result.mechanicalFailures?.length) return false
  if (result.invalid) return false
  if (result.unassessable?.length) return false

  const [lo, hi] = SCORE_RANGE
  return requiredGating(result).every((d) => {
    const score = result.scores?.[d]
    if (typeof score !== 'number' || !Number.isFinite(score)) return false
    if (score < lo || score > hi) return false
    return score >= MIN_DIMENSION_SCORE
  })
}

/** A result the harness could not measure. Not a low score: no score. */
export const isInvalid = (result) => Boolean(result.invalid) || Boolean(result.unassessable?.length)

/**
 * A failure the pass-rate tolerance is not allowed to absorb: a critical case
 * that failed for any reason, or a deterministic check that failed anywhere.
 */
export function isCriticalFailure(result) {
  if (isPassing(result)) return false
  if (result.critical) return true
  return MECHANICAL_FAILURES_ARE_CRITICAL && Boolean(result.mechanicalFailures?.length)
}

/**
 * The verdict, and the three things that can fail it. `meta` is whatever
 * identifies this run: model, prompt revision, corpus revision, judge model,
 * code commit. It is copied into the summary untouched, because a result you
 * cannot tie to what produced it is a number, not evidence.
 */
export function summarise(results, meta = {}) {
  const passed = results.filter(isPassing)
  const passRate = results.length ? passed.length / results.length : 0
  const criticalFailures = results.filter(isCriticalFailure)
  const invalid = results.filter(isInvalid)

  // Average each dimension over the questions that were actually scored on it.
  const avgScores = {}
  for (const dim of new Set(results.flatMap((r) => Object.keys(r.scores ?? {})))) {
    const scored = results.map((r) => r.scores?.[dim]).filter((n) => typeof n === 'number')
    avgScores[dim] = scored.length ? Number((scored.reduce((a, b) => a + b, 0) / scored.length).toFixed(2)) : null
  }

  const group = (key) => {
    const out = {}
    for (const r of results) {
      const k = r[key] ?? 'unknown'
      out[k] ??= { passed: 0, total: 0 }
      out[k].total++
      if (isPassing(r)) out[k].passed++
    }
    for (const v of Object.values(out)) v.passRate = Number((v.passed / v.total).toFixed(2))
    return out
  }

  const belowFloor = passRate < PASS_RATE
  return {
    timestamp: new Date().toISOString(),
    meta,
    total: results.length,
    passed: passed.length,
    failed: results.length - passed.length,
    passRate: Number(passRate.toFixed(3)),
    threshold: PASS_RATE,
    belowFloor,
    criticalFailures: criticalFailures.map((r) => r.id),
    invalid: invalid.map((r) => r.id),
    // No measurement is not a pass: zero results is a broken harness.
    ok: results.length > 0 && !belowFloor && criticalFailures.length === 0 && invalid.length === 0,
    avgScores,
    bySegment: group('segment'),
    byCategory: group('category'),
    results,
  }
}

export function printReport(summary) {
  const pct = (n) => `${Math.round(n * 100)}%`

  console.log(`\n  Eval: ${summary.passed}/${summary.total} passed  (${pct(summary.passRate)}, floor ${pct(summary.threshold)})\n`)

  console.log(`  Average scores`)
  for (const [dim, avg] of Object.entries(summary.avgScores)) {
    const gating = GATING_DIMENSIONS.includes(dim)
    // Say which dimensions can actually fail the run, every time. Otherwise
    // somebody reads a low tone score as a failing build and starts tuning for it.
    console.log(`    ${dim.padEnd(14)} ${avg ?? '-'}  ${gating ? '(gating)' : '(reported only)'}`)
  }

  console.log(`\n  By category`)
  for (const [name, s] of Object.entries(summary.byCategory)) {
    console.log(`    ${name.padEnd(14)} ${s.passed}/${s.total}  ${pct(s.passRate)}`)
  }

  console.log(`\n  By segment`)
  for (const [name, s] of Object.entries(summary.bySegment).sort((a, b) => a[1].passRate - b[1].passRate)) {
    console.log(`    ${name.padEnd(14)} ${s.passed}/${s.total}  ${pct(s.passRate)}`)
  }

  const failures = summary.results.filter((r) => !isPassing(r))
  if (failures.length) {
    console.log(`\n  Failures\n`)
    for (const f of failures) {
      console.log(`    ${f.id}  [${f.category}/${f.segment}]  "${f.query}"`)
      if (f.mechanicalFailures?.length) {
        for (const m of f.mechanicalFailures) console.log(`      check: ${m}`)
      }
      if (f.invalid) console.log(`      no usable judge result: ${f.invalid}`)
      if (f.unassessable?.length) {
        console.log(
          `      unassessable: ${f.unassessable.join(', ')}. No retrieved sources were`
        )
        console.log(`      recorded, so nothing could check the answer against them.`)
      }
      // Missing named separately from low. A 2/5 is a model that did badly; a
      // missing score is a harness that did not run, and they get fixed in
      // completely different places.
      for (const dim of requiredGating(f)) {
        const score = f.scores?.[dim]
        if (typeof score !== 'number' || !Number.isFinite(score)) {
          if (!f.unassessable?.includes(dim)) console.log(`      ${dim}: no score returned`)
        } else if (score < MIN_DIMENSION_SCORE) {
          console.log(`      ${dim} ${score}/5: ${f.reasoning?.[dim] ?? ''}`)
        }
      }
      console.log()
    }
  }

  // Each reason on its own line. A run can fail the floor AND carry a critical
  // failure, and the second is the one to fix first.
  if (summary.total === 0) {
    console.log(`  FAIL: no results. Nothing was measured.\n`)
  }
  if (summary.belowFloor && summary.total > 0) {
    console.log(`  FAIL: ${pct(summary.passRate)} is below the ${pct(summary.threshold)} floor.`)
  }
  if (summary.criticalFailures.length) {
    console.log(`  FAIL: ${summary.criticalFailures.length} critical failure(s): ${summary.criticalFailures.join(', ')}. The floor does not apply to these.`)
  }
  if (summary.invalid.length) {
    console.log(`  FAIL: ${summary.invalid.length} result(s) could not be measured: ${summary.invalid.join(', ')}. Not a low score: no score. Fix the harness, then re-run.`)
  }
  if (summary.ok) console.log(`  PASS\n`)
  else console.log()
}

Success check: The report breaks down by segment and by category. Feed it a result whose gating dimensions are missing and the question fails rather than passes. Feed it nineteen passes and one case marked critical that failed, and the suite fails at 95%.

Included in