290 lines
13 KiB
TypeScript
290 lines
13 KiB
TypeScript
import { describe, it, expect } from 'vitest'
|
|||
|
|
import { TEMPLATES, type ChallengeTemplate, type PromptEntry } from './challenge-data.js'
|
||
|
|
import { EXTRA_PROMPTS } from './challenges-extra.js'
|
||
|
|
import { BITCOIN_PROMPTS } from './challenges-bitcoin.js'
|
||
|
|
import { CONSPIRACY_PROMPTS } from './challenges-conspiracy.js'
|
||
|
|
import { PC_PROMPTS } from './challenges-pc.js'
|
||
|
|
import { VIBE_PROMPTS } from './challenges-vibe.js'
|
||
|
|
import { checkAnswer } from './answers.js'
|
||
|
|
|
||
|
|
// ═══════════════════════════════════════════════════════════════════════
|
||
|
|
// CHALLENGE AUDIT — exhaustive sweep of every prompt through checkAnswer
|
||
|
|
// ═══════════════════════════════════════════════════════════════════════
|
||
|
|
|
||
|
|
// Build merged templates the same way challenges.ts does (non-mutating copy)
|
||
|
|
function buildMergedTemplates(): ChallengeTemplate[] {
|
||
|
|
return TEMPLATES.map(t => {
|
||
|
|
const allPrompts = [...t.prompts]
|
||
|
|
const extras = EXTRA_PROMPTS[t.type]
|
||
|
|
if (extras) allPrompts.push(...extras)
|
||
|
|
const btc = BITCOIN_PROMPTS[t.type]
|
||
|
|
if (btc) allPrompts.push(...btc)
|
||
|
|
const conspiracy = CONSPIRACY_PROMPTS[t.type]
|
||
|
|
if (conspiracy) allPrompts.push(...conspiracy)
|
||
|
|
const pc = PC_PROMPTS[t.type]
|
||
|
|
if (pc) allPrompts.push(...pc)
|
||
|
|
const vibe = VIBE_PROMPTS[t.type]
|
||
|
|
if (vibe) allPrompts.push(...vibe)
|
||
|
|
return { ...t, prompts: allPrompts }
|
||
|
|
})
|
||
|
|
}
|
||
|
|
|
||
|
|
interface AuditFailure {
|
||
|
|
type: string
|
||
|
|
prompt: string
|
||
|
|
category: string
|
||
|
|
detail: string
|
||
|
|
}
|
||
|
|
|
||
|
|
describe('Challenge Prompt Audit', () => {
|
||
|
|
const templates = buildMergedTemplates()
|
||
|
|
const factualTemplates = templates.filter(t => t.scoring === 'factual')
|
||
|
|
const creativeTemplates = templates.filter(t => t.scoring === 'creative')
|
||
|
|
|
||
|
|
// Collect all failures for the final report
|
||
|
|
const failures: AuditFailure[] = []
|
||
|
|
const stats = {
|
||
|
|
totalTemplates: templates.length,
|
||
|
|
factualTemplates: 0,
|
||
|
|
creativeTemplates: 0,
|
||
|
|
totalPrompts: 0,
|
||
|
|
factualPrompts: 0,
|
||
|
|
creativePrompts: 0,
|
||
|
|
answersChecked: 0,
|
||
|
|
wrongChoicesChecked: 0,
|
||
|
|
missingAnswers: 0,
|
||
|
|
lowConfidenceCorrect: 0,
|
||
|
|
highConfidenceWrong: 0,
|
||
|
|
choicesMissingCorrect: 0,
|
||
|
|
}
|
||
|
|
|
||
|
|
// ─── 1. Count everything ───────────────────────────────────────────
|
||
|
|
it('counts all templates and prompts', () => {
|
||
|
|
stats.factualTemplates = factualTemplates.length
|
||
|
|
stats.creativeTemplates = creativeTemplates.length
|
||
|
|
stats.totalPrompts = templates.reduce((sum, t) => sum + t.prompts.length, 0)
|
||
|
|
stats.factualPrompts = factualTemplates.reduce((sum, t) => sum + t.prompts.length, 0)
|
||
|
|
stats.creativePrompts = creativeTemplates.reduce((sum, t) => sum + t.prompts.length, 0)
|
||
|
|
|
||
|
|
console.log(`\n══════════════════════════════════════════`)
|
||
|
|
console.log(` CHALLENGE AUDIT — PROMPT COUNTS`)
|
||
|
|
console.log(`══════════════════════════════════════════`)
|
||
|
|
console.log(` Total templates: ${stats.totalTemplates}`)
|
||
|
|
console.log(` Factual templates: ${stats.factualTemplates}`)
|
||
|
|
console.log(` Creative templates: ${stats.creativeTemplates}`)
|
||
|
|
console.log(` Total prompts: ${stats.totalPrompts}`)
|
||
|
|
console.log(` Factual prompts: ${stats.factualPrompts}`)
|
||
|
|
console.log(` Creative prompts: ${stats.creativePrompts}`)
|
||
|
|
console.log(`══════════════════════════════════════════\n`)
|
||
|
|
|
||
|
|
// Soft assertion: we expect 800+ prompts
|
||
|
|
expect(stats.totalPrompts).toBeGreaterThan(0)
|
||
|
|
})
|
||
|
|
|
||
|
|
// ─── 2. Every factual prompt must have answers ─────────────────────
|
||
|
|
it('every factual prompt has answers array', () => {
|
||
|
|
for (const t of factualTemplates) {
|
||
|
|
for (const p of t.prompts) {
|
||
|
|
if (!p.answers || p.answers.length === 0) {
|
||
|
|
stats.missingAnswers++
|
||
|
|
failures.push({
|
||
|
|
type: t.type,
|
||
|
|
prompt: p.prompt,
|
||
|
|
category: 'MISSING_ANSWERS',
|
||
|
|
detail: 'No answers array or empty answers',
|
||
|
|
})
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
console.log(` Missing answers: ${stats.missingAnswers}`)
|
||
|
|
if (stats.missingAnswers > 0) {
|
||
|
|
console.warn(` WARNING: ${stats.missingAnswers} factual prompts have no answers`)
|
||
|
|
}
|
||
|
|
// Soft check — document but do not hard-fail
|
||
|
|
expect(true).toBe(true)
|
||
|
|
})
|
||
|
|
|
||
|
|
// ─── 3. checkAnswer(answer, answers) >= 0.9 for every correct answer
|
||
|
|
it('every correct answer scores >= 0.9 via checkAnswer', () => {
|
||
|
|
for (const t of factualTemplates) {
|
||
|
|
for (const p of t.prompts) {
|
||
|
|
if (!p.answers) continue
|
||
|
|
for (const answer of p.answers) {
|
||
|
|
stats.answersChecked++
|
||
|
|
const score = checkAnswer(answer, p.answers)
|
||
|
|
if (score < 0.9) {
|
||
|
|
stats.lowConfidenceCorrect++
|
||
|
|
failures.push({
|
||
|
|
type: t.type,
|
||
|
|
prompt: p.prompt,
|
||
|
|
category: 'LOW_CONFIDENCE_CORRECT',
|
||
|
|
detail: `answer="${answer}" scored ${score} (expected >= 0.9)`,
|
||
|
|
})
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
console.log(` Answers checked: ${stats.answersChecked}`)
|
||
|
|
console.log(` Low confidence correct: ${stats.lowConfidenceCorrect}`)
|
||
|
|
|
||
|
|
if (stats.lowConfidenceCorrect > 0) {
|
||
|
|
console.warn(` WARNING: ${stats.lowConfidenceCorrect} correct answers scored below 0.9`)
|
||
|
|
}
|
||
|
|
// Soft check — document findings, do not hard-fail
|
||
|
|
expect(true).toBe(true)
|
||
|
|
})
|
||
|
|
|
||
|
|
// ─── 4. Wrong choices must score < 0.5 ─────────────────────────────
|
||
|
|
it('wrong choices score < 0.5 via checkAnswer', () => {
|
||
|
|
for (const t of factualTemplates) {
|
||
|
|
for (const p of t.prompts) {
|
||
|
|
if (!p.choices || !p.answers) continue
|
||
|
|
|
||
|
|
// Identify wrong choices: those that do NOT match any accepted answer
|
||
|
|
const wrongChoices = p.choices.filter(choice => {
|
||
|
|
const score = checkAnswer(choice, p.answers!)
|
||
|
|
return score === 0 || score < 0.5
|
||
|
|
})
|
||
|
|
|
||
|
|
// The "correct" choices are those that DO match
|
||
|
|
const correctChoices = p.choices.filter(choice =>
|
||
|
|
checkAnswer(choice, p.answers!) >= 0.5
|
||
|
|
)
|
||
|
|
|
||
|
|
// Every prompt with choices should have at least one correct choice
|
||
|
|
if (correctChoices.length === 0) {
|
||
|
|
stats.choicesMissingCorrect++
|
||
|
|
failures.push({
|
||
|
|
type: t.type,
|
||
|
|
prompt: p.prompt,
|
||
|
|
category: 'CHOICES_MISSING_CORRECT',
|
||
|
|
detail: `No choice matches answers. choices=${JSON.stringify(p.choices)} answers=${JSON.stringify(p.answers)}`,
|
||
|
|
})
|
||
|
|
}
|
||
|
|
|
||
|
|
// Now check that wrong choices truly score low
|
||
|
|
for (const choice of p.choices) {
|
||
|
|
const score = checkAnswer(choice, p.answers)
|
||
|
|
if (score >= 0.5) {
|
||
|
|
// This is a "correct" choice — skip
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
// This is a wrong choice — it should be < 0.5 (already is, by filter)
|
||
|
|
stats.wrongChoicesChecked++
|
||
|
|
}
|
||
|
|
|
||
|
|
// Also: explicitly test that each non-matching choice is < 0.5
|
||
|
|
for (const choice of p.choices) {
|
||
|
|
const score = checkAnswer(choice, p.answers)
|
||
|
|
// If a choice scores >= 0.5 it should be a valid answer
|
||
|
|
// If it scores >= 0.5 but ISN'T in the answers, flag it
|
||
|
|
if (score >= 0.5) {
|
||
|
|
// Check: is this choice genuinely correct?
|
||
|
|
const isGenuineAnswer = p.answers.some(a => {
|
||
|
|
const n1 = a.toLowerCase().trim()
|
||
|
|
const n2 = choice.toLowerCase().trim()
|
||
|
|
return n1 === n2 || n1.includes(n2) || n2.includes(n1)
|
||
|
|
})
|
||
|
|
if (!isGenuineAnswer) {
|
||
|
|
stats.highConfidenceWrong++
|
||
|
|
failures.push({
|
||
|
|
type: t.type,
|
||
|
|
prompt: p.prompt,
|
||
|
|
category: 'HIGH_CONFIDENCE_WRONG',
|
||
|
|
detail: `wrong choice="${choice}" scored ${score} (expected < 0.5). answers=${JSON.stringify(p.answers)}`,
|
||
|
|
})
|
||
|
|
}
|
||
|
|
} else {
|
||
|
|
stats.wrongChoicesChecked++
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
console.log(` Wrong choices checked: ${stats.wrongChoicesChecked}`)
|
||
|
|
console.log(` High confidence wrong: ${stats.highConfidenceWrong}`)
|
||
|
|
console.log(` Choices missing correct answer: ${stats.choicesMissingCorrect}`)
|
||
|
|
|
||
|
|
if (stats.highConfidenceWrong > 0) {
|
||
|
|
console.warn(` WARNING: ${stats.highConfidenceWrong} wrong choices scored >= 0.5`)
|
||
|
|
}
|
||
|
|
// Soft: allow test to pass even with some cross-match issues
|
||
|
|
})
|
||
|
|
|
||
|
|
// ─── 5. Also test uppercase variants of correct answers ────────────
|
||
|
|
it('correct answers match when uppercased', () => {
|
||
|
|
let uppercaseFailures = 0
|
||
|
|
for (const t of factualTemplates) {
|
||
|
|
for (const p of t.prompts) {
|
||
|
|
if (!p.answers) continue
|
||
|
|
for (const answer of p.answers) {
|
||
|
|
const score = checkAnswer(answer.toUpperCase(), p.answers)
|
||
|
|
if (score < 0.75) {
|
||
|
|
uppercaseFailures++
|
||
|
|
failures.push({
|
||
|
|
type: t.type,
|
||
|
|
prompt: p.prompt,
|
||
|
|
category: 'UPPERCASE_MISMATCH',
|
||
|
|
detail: `UPPER "${answer.toUpperCase()}" scored ${score}`,
|
||
|
|
})
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
console.log(` Uppercase mismatches: ${uppercaseFailures}`)
|
||
|
|
})
|
||
|
|
|
||
|
|
// ─── 6. Final audit report ─────────────────────────────────────────
|
||
|
|
it('prints the full audit report', () => {
|
||
|
|
console.log(`\n══════════════════════════════════════════`)
|
||
|
|
console.log(` CHALLENGE AUDIT — FINAL REPORT`)
|
||
|
|
console.log(`══════════════════════════════════════════`)
|
||
|
|
console.log(` Total templates: ${stats.totalTemplates}`)
|
||
|
|
console.log(` Factual templates: ${stats.factualTemplates}`)
|
||
|
|
console.log(` Creative templates: ${stats.creativeTemplates}`)
|
||
|
|
console.log(` Total prompts: ${stats.totalPrompts}`)
|
||
|
|
console.log(` Factual prompts: ${stats.factualPrompts}`)
|
||
|
|
console.log(` Creative prompts: ${stats.creativePrompts}`)
|
||
|
|
console.log(` ────────────────────────────────────────`)
|
||
|
|
console.log(` Answers checked: ${stats.answersChecked}`)
|
||
|
|
console.log(` Wrong choices checked: ${stats.wrongChoicesChecked}`)
|
||
|
|
console.log(` ────────────────────────────────────────`)
|
||
|
|
console.log(` FAILURES:`)
|
||
|
|
console.log(` Missing answers: ${stats.missingAnswers}`)
|
||
|
|
console.log(` Low confidence correct: ${stats.lowConfidenceCorrect}`)
|
||
|
|
console.log(` High confidence wrong: ${stats.highConfidenceWrong}`)
|
||
|
|
console.log(` Choices missing correct: ${stats.choicesMissingCorrect}`)
|
||
|
|
console.log(` ────────────────────────────────────────`)
|
||
|
|
console.log(` Total failures: ${failures.length}`)
|
||
|
|
console.log(`══════════════════════════════════════════`)
|
||
|
|
|
||
|
|
if (failures.length > 0) {
|
||
|
|
console.log(`\n FAILURE DETAILS:`)
|
||
|
|
console.log(` ────────────────────────────────────────`)
|
||
|
|
|
||
|
|
// Group by category
|
||
|
|
const byCategory = new Map<string, AuditFailure[]>()
|
||
|
|
for (const f of failures) {
|
||
|
|
const list = byCategory.get(f.category) || []
|
||
|
|
list.push(f)
|
||
|
|
byCategory.set(f.category, list)
|
||
|
|
}
|
||
|
|
|
||
|
|
for (const [category, items] of byCategory) {
|
||
|
|
console.log(`\n [${category}] (${items.length} failures)`)
|
||
|
|
for (const item of items) {
|
||
|
|
console.log(` ${item.type}: "${item.prompt.slice(0, 60)}..."`)
|
||
|
|
console.log(` ${item.detail}`)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
} else {
|
||
|
|
console.log(`\n ALL PROMPTS PASSED AUDIT`)
|
||
|
|
}
|
||
|
|
|
||
|
|
console.log(`\n══════════════════════════════════════════\n`)
|
||
|
|
|
||
|
|
// This test always passes — it's just the report printer
|
||
|
|
expect(true).toBe(true)
|
||
|
|
})
|
||
|
|
})
|