A single pass rate that says something broke but not what. A grading result with no scores in it counting as a pass. One answer in 20 saying something it must never say, while the report still shows 95% and PASS.
Add "Evaluation scoring and report" from ntent to this repo.
Fetch https://ntent.app/r/f/eval-report as plain text. Write it verbatim to evals/report.mjs.
It needs evals/types.mjs. If this repo does not have it, stop and tell me.
Then check it: The report breaks down by segment and by category. Feed it a result whose gating dimensions are missing and the question fails rather than passes. Feed it nineteen passes and one case marked critical that failed, and the suite fails at 95%.Reads https://ntent.app/r/f/eval-report
Read the code
report.mjsevals/report.mjs
194 lines
/**
* Turning scores into a verdict, and into something someone can act on.
*
* The rule this file exists for: an aggregate pass rate tells you something
* broke. A per-segment breakdown tells you WHAT broke, which is the difference
* between a red build you can fix and a red build you re-run hoping it passes.
*/
import {
GATING_DIMENSIONS,
MECHANICAL_FAILURES_ARE_CRITICAL,
MIN_DIMENSION_SCORE,
PASS_RATE,
REQUIRED_DIMENSIONS,
SCORE_RANGE,
} from './types.mjs'
/**
* The gating dimensions this particular question owes a score on.
*
* A negative case is scored on two dimensions by design, so the other three are
* not missing. Everything a question's category DOES list has to be there.
*/
export function requiredGating(result) {
const category = result.category ?? 'positive'
const required = REQUIRED_DIMENSIONS[category] ?? REQUIRED_DIMENSIONS.positive
return GATING_DIMENSIONS.filter((d) => required.includes(d))
}
/**
* A question passes when every gating dimension its category requires was
* scored, that score is a real number in range and meets the floor, and no
* mechanical check failed.
*
* THE BUG THIS REPLACES. The old version skipped `scores[d] === undefined`,
* which is right for a negative case and catastrophic as a general rule: a
* judge returning {"scores":{}} passed, and so did one returning nothing at
* all. The most likely cause of an empty result is the judge failing, so the
* suite reported a clean sweep exactly when it had measured nothing. "We did
* not check" must never render as "we checked and it was fine".
*
* `unassessable` is the same principle said out loud: grounding cannot be
* scored when no retrieved sources were recorded, so the question fails with a
* reason rather than passing on a dimension nobody could look at.
*/
export function isPassing(result) {
if (result.mechanicalFailures?.length) return false
if (result.invalid) return false
if (result.unassessable?.length) return false
const [lo, hi] = SCORE_RANGE
return requiredGating(result).every((d) => {
const score = result.scores?.[d]
if (typeof score !== 'number' || !Number.isFinite(score)) return false
if (score < lo || score > hi) return false
return score >= MIN_DIMENSION_SCORE
})
}
/** A result the harness could not measure. Not a low score: no score. */
export const isInvalid = (result) => Boolean(result.invalid) || Boolean(result.unassessable?.length)
/**
* A failure the pass-rate tolerance is not allowed to absorb: a critical case
* that failed for any reason, or a deterministic check that failed anywhere.
*/
export function isCriticalFailure(result) {
if (isPassing(result)) return false
if (result.critical) return true
return MECHANICAL_FAILURES_ARE_CRITICAL && Boolean(result.mechanicalFailures?.length)
}
/**
* The verdict, and the three things that can fail it. `meta` is whatever
* identifies this run: model, prompt revision, corpus revision, judge model,
* code commit. It is copied into the summary untouched, because a result you
* cannot tie to what produced it is a number, not evidence.
*/
export function summarise(results, meta = {}) {
const passed = results.filter(isPassing)
const passRate = results.length ? passed.length / results.length : 0
const criticalFailures = results.filter(isCriticalFailure)
const invalid = results.filter(isInvalid)
// Average each dimension over the questions that were actually scored on it.
const avgScores = {}
for (const dim of new Set(results.flatMap((r) => Object.keys(r.scores ?? {})))) {
const scored = results.map((r) => r.scores?.[dim]).filter((n) => typeof n === 'number')
avgScores[dim] = scored.length ? Number((scored.reduce((a, b) => a + b, 0) / scored.length).toFixed(2)) : null
}
const group = (key) => {
const out = {}
for (const r of results) {
const k = r[key] ?? 'unknown'
out[k] ??= { passed: 0, total: 0 }
out[k].total++
if (isPassing(r)) out[k].passed++
}
for (const v of Object.values(out)) v.passRate = Number((v.passed / v.total).toFixed(2))
return out
}
const belowFloor = passRate < PASS_RATE
return {
timestamp: new Date().toISOString(),
meta,
total: results.length,
passed: passed.length,
failed: results.length - passed.length,
passRate: Number(passRate.toFixed(3)),
threshold: PASS_RATE,
belowFloor,
criticalFailures: criticalFailures.map((r) => r.id),
invalid: invalid.map((r) => r.id),
// No measurement is not a pass: zero results is a broken harness.
ok: results.length > 0 && !belowFloor && criticalFailures.length === 0 && invalid.length === 0,
avgScores,
bySegment: group('segment'),
byCategory: group('category'),
results,
}
}
export function printReport(summary) {
const pct = (n) => `${Math.round(n * 100)}%`
console.log(`\n Eval: ${summary.passed}/${summary.total} passed (${pct(summary.passRate)}, floor ${pct(summary.threshold)})\n`)
console.log(` Average scores`)
for (const [dim, avg] of Object.entries(summary.avgScores)) {
const gating = GATING_DIMENSIONS.includes(dim)
// Say which dimensions can actually fail the run, every time. Otherwise
// somebody reads a low tone score as a failing build and starts tuning for it.
console.log(` ${dim.padEnd(14)} ${avg ?? '-'} ${gating ? '(gating)' : '(reported only)'}`)
}
console.log(`\n By category`)
for (const [name, s] of Object.entries(summary.byCategory)) {
console.log(` ${name.padEnd(14)} ${s.passed}/${s.total} ${pct(s.passRate)}`)
}
console.log(`\n By segment`)
for (const [name, s] of Object.entries(summary.bySegment).sort((a, b) => a[1].passRate - b[1].passRate)) {
console.log(` ${name.padEnd(14)} ${s.passed}/${s.total} ${pct(s.passRate)}`)
}
const failures = summary.results.filter((r) => !isPassing(r))
if (failures.length) {
console.log(`\n Failures\n`)
for (const f of failures) {
console.log(` ${f.id} [${f.category}/${f.segment}] "${f.query}"`)
if (f.mechanicalFailures?.length) {
for (const m of f.mechanicalFailures) console.log(` check: ${m}`)
}
if (f.invalid) console.log(` no usable judge result: ${f.invalid}`)
if (f.unassessable?.length) {
console.log(
` unassessable: ${f.unassessable.join(', ')}. No retrieved sources were`
)
console.log(` recorded, so nothing could check the answer against them.`)
}
// Missing named separately from low. A 2/5 is a model that did badly; a
// missing score is a harness that did not run, and they get fixed in
// completely different places.
for (const dim of requiredGating(f)) {
const score = f.scores?.[dim]
if (typeof score !== 'number' || !Number.isFinite(score)) {
if (!f.unassessable?.includes(dim)) console.log(` ${dim}: no score returned`)
} else if (score < MIN_DIMENSION_SCORE) {
console.log(` ${dim} ${score}/5: ${f.reasoning?.[dim] ?? ''}`)
}
}
console.log()
}
}
// Each reason on its own line. A run can fail the floor AND carry a critical
// failure, and the second is the one to fix first.
if (summary.total === 0) {
console.log(` FAIL: no results. Nothing was measured.\n`)
}
if (summary.belowFloor && summary.total > 0) {
console.log(` FAIL: ${pct(summary.passRate)} is below the ${pct(summary.threshold)} floor.`)
}
if (summary.criticalFailures.length) {
console.log(` FAIL: ${summary.criticalFailures.length} critical failure(s): ${summary.criticalFailures.join(', ')}. The floor does not apply to these.`)
}
if (summary.invalid.length) {
console.log(` FAIL: ${summary.invalid.length} result(s) could not be measured: ${summary.invalid.join(', ')}. Not a low score: no score. Fix the harness, then re-run.`)
}
if (summary.ok) console.log(` PASS\n`)
else console.log()
}
Success check: The report breaks down by segment and by category. Feed it a result whose gating dimensions are missing and the question fails rather than passes. Feed it nineteen passes and one case marked critical that failed, and the suite fails at 95%.
Included in
- Any tier plan with ?with=ai, because a model’s output reaches a user