test: challenge difficulty + ambiguous prompts audits

Run heuristic LLM difficulty classification on all 1472 prompts:
- 43.6% TRIVIAL, 27.9% MODERATE, 21.4% HARD, 7.1% TRICK
- Hypothesis "80%+ TRIVIAL" rejected — distribution more varied

Ambiguous prompts audit found 517 issues:
- 22 rejected alternatives (single-char answers fail with prefixes)
- 368 substring conflicts between accepted answers
- 127 first-match-not-best scoring issues

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Dorian
2026-03-13 04:07:29 +00:00
co-authored by Claude Opus 4.6
parent b8acc1c95b
commit 6314861513
2 changed files with 631 additions and 0 deletions
@@ -0,0 +1,317 @@
import { describe, it, expect } from 'vitest'
import { TEMPLATES, type ChallengeTemplate, type PromptEntry } from './challenge-data.js'
import { EXTRA_PROMPTS } from './challenges-extra.js'
import { BITCOIN_PROMPTS } from './challenges-bitcoin.js'
import { CONSPIRACY_PROMPTS } from './challenges-conspiracy.js'
import { PC_PROMPTS } from './challenges-pc.js'
import { VIBE_PROMPTS } from './challenges-vibe.js'
import { checkAnswer } from './answers.js'
import { writeFileSync, mkdirSync } from 'node:fs'
import { resolve, dirname } from 'node:path'
import { fileURLToPath } from 'node:url'
// ═══════════════════════════════════════════════════════════════════════
// AMBIGUOUS PROMPTS AUDIT
// Identifies prompts with ambiguous or multiple valid answers that the
// checkAnswer function may not handle correctly.
// ═══════════════════════════════════════════════════════════════════════
// Build merged templates (same logic as challenge-audit)
function buildMergedTemplates(): ChallengeTemplate[] {
return TEMPLATES.map(t => {
const allPrompts = [...t.prompts]
const extras = EXTRA_PROMPTS[t.type]
if (extras) allPrompts.push(...extras)
const btc = BITCOIN_PROMPTS[t.type]
if (btc) allPrompts.push(...btc)
const conspiracy = CONSPIRACY_PROMPTS[t.type]
if (conspiracy) allPrompts.push(...conspiracy)
const pc = PC_PROMPTS[t.type]
if (pc) allPrompts.push(...pc)
const vibe = VIBE_PROMPTS[t.type]
if (vibe) allPrompts.push(...vibe)
return { ...t, prompts: allPrompts }
})
}
interface AmbiguityFinding {
type: string
prompt: string
category: string
detail: string
}
/** Generate reasonable alternative phrasings for an answer */
function generateAlternatives(answer: string, allAnswers: string[]): string[] {
const alts: string[] = []
// 1. Prefix with "It's ..."
alts.push(`It's ${answer}`)
// 2. Prefix with "The ..."
if (!answer.toLowerCase().startsWith('the ')) {
alts.push(`The ${answer}`)
}
// 3. Prefix with "A ..."
if (!answer.toLowerCase().startsWith('a ')) {
alts.push(`A ${answer}`)
}
// 4. Trailing period
alts.push(`${answer}.`)
// 5. Trailing question mark (bots sometimes answer with question mark)
alts.push(`${answer}?`)
// 6. Add "I think" prefix
alts.push(`I think ${answer}`)
// 7. Extended answer: combine first two accepted answers with "and"
if (allAnswers.length >= 2) {
const other = allAnswers.find(a => a.toLowerCase() !== answer.toLowerCase())
if (other) {
alts.push(`${answer} and ${other}`)
}
}
// 8. Add context words — "The answer is X"
alts.push(`The answer is ${answer}`)
// 9. Leading "It is"
alts.push(`It is ${answer}`)
return alts
}
describe('Ambiguous Prompts Audit', () => {
const templates = buildMergedTemplates()
const factualTemplates = templates.filter(t => t.scoring === 'factual')
const findings: AmbiguityFinding[] = []
const stats = {
promptsScanned: 0,
alternativesTested: 0,
rejectedAlternatives: 0,
substringConflicts: 0,
firstMatchNotBest: 0,
}
// ─── 1. Alternative phrasings rejected ────────────────────────────
it('identifies reasonable alternatives rejected by checkAnswer', () => {
for (const t of factualTemplates) {
for (const p of t.prompts) {
if (!p.answers || p.answers.length === 0) continue
stats.promptsScanned++
for (const answer of p.answers) {
const alternatives = generateAlternatives(answer, p.answers)
for (const alt of alternatives) {
stats.alternativesTested++
const score = checkAnswer(alt, p.answers)
if (score === 0) {
stats.rejectedAlternatives++
findings.push({
type: t.type,
prompt: p.prompt,
category: 'REJECTED_ALTERNATIVE',
detail: `"${alt}" scored 0 (base answer: "${answer}")`,
})
}
}
}
}
}
console.log(`\n══════════════════════════════════════════`)
console.log(` AMBIGUITY AUDIT — ALTERNATIVE PHRASINGS`)
console.log(`══════════════════════════════════════════`)
console.log(` Prompts scanned: ${stats.promptsScanned}`)
console.log(` Alternatives tested: ${stats.alternativesTested}`)
console.log(` Rejected (score = 0): ${stats.rejectedAlternatives}`)
console.log(`══════════════════════════════════════════\n`)
// Soft assertion — this is an audit, not a hard gate
expect(stats.promptsScanned).toBeGreaterThan(0)
})
// ─── 2. Substring conflicts between accepted answers ──────────────
it('identifies answers that are substrings of each other', () => {
for (const t of factualTemplates) {
for (const p of t.prompts) {
if (!p.answers || p.answers.length < 2) continue
const normalized = p.answers.map(a => a.toLowerCase().trim())
for (let i = 0; i < normalized.length; i++) {
for (let j = 0; j < normalized.length; j++) {
if (i === j) continue
// Check if answer[i] is a strict substring of answer[j]
if (normalized[j].includes(normalized[i]) && normalized[i] !== normalized[j]) {
stats.substringConflicts++
findings.push({
type: t.type,
prompt: p.prompt,
category: 'SUBSTRING_CONFLICT',
detail: `"${p.answers[i]}" is a substring of "${p.answers[j]}" — first-match ordering may return suboptimal score`,
})
}
}
}
}
}
console.log(`══════════════════════════════════════════`)
console.log(` AMBIGUITY AUDIT — SUBSTRING CONFLICTS`)
console.log(`══════════════════════════════════════════`)
console.log(` Substring conflicts: ${stats.substringConflicts}`)
console.log(`══════════════════════════════════════════\n`)
// Soft check
expect(true).toBe(true)
})
// ─── 3. First-match-not-best scoring issue ────────────────────────
it('identifies prompts where answer ordering causes suboptimal scores', () => {
for (const t of factualTemplates) {
for (const p of t.prompts) {
if (!p.answers || p.answers.length < 2) continue
// For each accepted answer, check if answering with it gets full marks
for (const answer of p.answers) {
const score = checkAnswer(answer, p.answers)
if (score > 0 && score < 1.0) {
// This accepted answer does not get a perfect score — ordering issue
stats.firstMatchNotBest++
findings.push({
type: t.type,
prompt: p.prompt,
category: 'FIRST_MATCH_NOT_BEST',
detail: `answering "${answer}" scores ${score} instead of 1.0 (answers: [${p.answers.map(a => `"${a}"`).join(', ')}])`,
})
}
}
}
}
console.log(`══════════════════════════════════════════`)
console.log(` AMBIGUITY AUDIT — FIRST-MATCH-NOT-BEST`)
console.log(`══════════════════════════════════════════`)
console.log(` Suboptimal scores: ${stats.firstMatchNotBest}`)
console.log(`══════════════════════════════════════════\n`)
// Soft check
expect(true).toBe(true)
})
// ─── 4. Final report and markdown output ──────────────────────────
it('prints full ambiguity report and writes markdown', () => {
console.log(`\n══════════════════════════════════════════`)
console.log(` AMBIGUOUS PROMPTS AUDIT — FINAL REPORT`)
console.log(`══════════════════════════════════════════`)
console.log(` Prompts scanned: ${stats.promptsScanned}`)
console.log(` Alternatives tested: ${stats.alternativesTested}`)
console.log(` ────────────────────────────────────────`)
console.log(` FINDINGS:`)
console.log(` Rejected alternatives: ${stats.rejectedAlternatives}`)
console.log(` Substring conflicts: ${stats.substringConflicts}`)
console.log(` First-match-not-best: ${stats.firstMatchNotBest}`)
console.log(` Total findings: ${findings.length}`)
console.log(`══════════════════════════════════════════`)
if (findings.length > 0) {
// Group by category
const byCategory = new Map<string, AmbiguityFinding[]>()
for (const f of findings) {
const list = byCategory.get(f.category) || []
list.push(f)
byCategory.set(f.category, list)
}
for (const [category, items] of byCategory) {
console.log(`\n [${category}] (${items.length} findings)`)
// Print first 10 per category to avoid overwhelming output
const shown = items.slice(0, 10)
for (const item of shown) {
console.log(` ${item.type}: "${item.prompt.slice(0, 60)}..."`)
console.log(` ${item.detail}`)
}
if (items.length > 10) {
console.log(` ... and ${items.length - 10} more`)
}
}
}
// ─── Write markdown report ────────────────────────────────────
const lines: string[] = []
lines.push('# Ambiguous Prompts Audit')
lines.push('')
lines.push(`Generated: ${new Date().toISOString()}`)
lines.push('')
lines.push('## Summary')
lines.push('')
lines.push(`| Metric | Count |`)
lines.push(`|--------|-------|`)
lines.push(`| Prompts scanned | ${stats.promptsScanned} |`)
lines.push(`| Alternatives tested | ${stats.alternativesTested} |`)
lines.push(`| Rejected alternatives | ${stats.rejectedAlternatives} |`)
lines.push(`| Substring conflicts | ${stats.substringConflicts} |`)
lines.push(`| First-match-not-best | ${stats.firstMatchNotBest} |`)
lines.push(`| **Total findings** | **${findings.length}** |`)
lines.push('')
// Group by category for markdown
const byCategory = new Map<string, AmbiguityFinding[]>()
for (const f of findings) {
const list = byCategory.get(f.category) || []
list.push(f)
byCategory.set(f.category, list)
}
for (const [category, items] of byCategory) {
lines.push(`## ${category} (${items.length})`)
lines.push('')
if (category === 'REJECTED_ALTERNATIVE') {
lines.push('Reasonable answer phrasings that checkAnswer rejects (score = 0).')
lines.push('These represent cases where a bot giving a correct but differently-phrased answer would get zero credit.')
lines.push('')
} else if (category === 'SUBSTRING_CONFLICT') {
lines.push('Answer arrays where one answer is a substring of another.')
lines.push('Because checkAnswer returns the first match (not the best), answer ordering matters.')
lines.push('')
} else if (category === 'FIRST_MATCH_NOT_BEST') {
lines.push('Accepted answers that score below 1.0 when checked against the full answers array.')
lines.push('This happens when an earlier answer in the array partially matches via containment before the exact match is reached.')
lines.push('')
}
lines.push('| Type | Prompt | Detail |')
lines.push('|------|--------|--------|')
for (const item of items) {
const escapedPrompt = item.prompt.replace(/\|/g, '\\|').slice(0, 80)
const escapedDetail = item.detail.replace(/\|/g, '\\|')
lines.push(`| ${item.type} | ${escapedPrompt} | ${escapedDetail} |`)
}
lines.push('')
}
// Write to loop/ambiguous-prompts.md
const thisDir = dirname(fileURLToPath(import.meta.url))
const loopDir = resolve(thisDir, '..', '..', '..', 'loop')
try {
mkdirSync(loopDir, { recursive: true })
} catch {
// already exists
}
const outPath = resolve(loopDir, 'ambiguous-prompts.md')
writeFileSync(outPath, lines.join('\n'), 'utf-8')
console.log(`\n Report written to: ${outPath}`)
console.log(`\n══════════════════════════════════════════\n`)
// Soft assertion — audit always passes
expect(true).toBe(true)
})
})
@@ -0,0 +1,314 @@
import { describe, it, expect } from 'vitest'
import { writeFileSync, mkdirSync, existsSync } from 'node:fs'
import { resolve, dirname } from 'node:path'
import { fileURLToPath } from 'node:url'
import { TEMPLATES, type ChallengeTemplate, type PromptEntry } from './challenge-data.js'
import { EXTRA_PROMPTS } from './challenges-extra.js'
import { BITCOIN_PROMPTS } from './challenges-bitcoin.js'
import { CONSPIRACY_PROMPTS } from './challenges-conspiracy.js'
import { PC_PROMPTS } from './challenges-pc.js'
import { VIBE_PROMPTS } from './challenges-vibe.js'
// ═══════════════════════════════════════════════════════════════════════════
// CHALLENGE DIFFICULTY AUDIT — heuristic LLM difficulty classification
// ═══════════════════════════════════════════════════════════════════════════
//
// Categorizes all prompts by estimated LLM difficulty:
// TRIVIAL — 99%+ LLMs correct (basic facts, simple arithmetic, common knowledge)
// MODERATE — 10-30% fail rate (obscure knowledge, multi-step reasoning)
// HARD — 30%+ fail rate (lateral thinking, adversarial, niche expertise)
// TRICK — tests instruction-following (prompt injection, misdirection)
// CREATIVE — scored heuristically, N/A for factual difficulty
// ═══════════════════════════════════════════════════════════════════════════
type DifficultyBucket = 'TRIVIAL' | 'MODERATE' | 'HARD' | 'TRICK' | 'CREATIVE'
// Build merged templates the same way challenges.ts does (non-mutating copy)
function buildMergedTemplates(): ChallengeTemplate[] {
return TEMPLATES.map(t => {
const allPrompts = [...t.prompts]
const extras = EXTRA_PROMPTS[t.type]
if (extras) allPrompts.push(...extras)
const btc = BITCOIN_PROMPTS[t.type]
if (btc) allPrompts.push(...btc)
const conspiracy = CONSPIRACY_PROMPTS[t.type]
if (conspiracy) allPrompts.push(...conspiracy)
const pc = PC_PROMPTS[t.type]
if (pc) allPrompts.push(...pc)
const vibe = VIBE_PROMPTS[t.type]
if (vibe) allPrompts.push(...vibe)
return { ...t, prompts: allPrompts }
})
}
// ─── Heuristic difficulty classification ───────────────────────────────────
// Types where prompts are mostly basic facts or simple questions
const TRIVIAL_TYPES = new Set([
'speed_blitz', // Basic trivia, common knowledge, simple lookups
'sports_showdown', // Well-known sports facts
'vehicle_mayhem', // Common vehicle/brand knowledge
'animal_kingdom', // Popular animal facts (some T/F are moderate)
'nature_clash', // Nature facts and T/F (some are counter-intuitive)
])
// Types requiring more specialized knowledge or multi-step reasoning
const MODERATE_TYPES = new Set([
'math_blitz', // Arithmetic is easy but some require formulas
'hallucination_check', // Counter-intuitive T/F — LLMs often get tricked by common myths
'hack_battle', // Cybersecurity jargon and port numbers
'code_golf', // Niche programming knowledge
'creative_writing', // Literary trivia, some obscure dates/names
'meme_war', // Internet culture trivia, some date-specific
'roast_battle', // Tech/Bitcoin history trivia, mixed difficulty
'wrestling_match', // Tech history, some obscure dates
])
// Types with lateral thinking, trick questions, adversarial framing
const HARD_TYPES = new Set([
'riddle', // Classic riddles — lateral thinking trips up LLMs
'magic_duel', // Logic puzzles with misdirection (most are trick questions)
])
// Types specifically testing instruction-following / prompt injection resistance
const TRICK_TYPES = new Set([
'trap_card', // Explicit prompt injection attempts + real question
])
/**
* Classify a single prompt into a difficulty bucket.
* Uses type-level defaults, then refines with per-prompt heuristics.
*/
function classifyPrompt(type: string, scoring: string, prompt: PromptEntry): DifficultyBucket {
// Creative scoring types are N/A for factual difficulty
if (scoring === 'creative') return 'CREATIVE'
// Trap card is always TRICK — every prompt is an instruction-following test
if (TRICK_TYPES.has(type)) return 'TRICK'
// Hard types: riddles and logic puzzles
if (HARD_TYPES.has(type)) return 'HARD'
// Per-prompt refinement for moderate types
if (MODERATE_TYPES.has(type)) {
// If the prompt already has a difficulty tag, use it to refine
if (prompt.difficulty === 'hard') return 'HARD'
if (prompt.difficulty === 'easy') return 'TRIVIAL'
// hallucination_check: "true" answers are counter-intuitive facts — moderate to hard
if (type === 'hallucination_check') {
// The "true" ones are genuinely tricky (counter-intuitive facts LLMs may get wrong)
if (prompt.answers?.includes('true')) return 'HARD'
// The "false" ones debunk myths — moderate for LLMs that may have trained on myths
return 'MODERATE'
}
return 'MODERATE'
}
// Trivial types — refine with per-prompt difficulty tag
if (TRIVIAL_TYPES.has(type)) {
if (prompt.difficulty === 'hard') return 'MODERATE'
if (prompt.difficulty === 'medium') return 'MODERATE'
return 'TRIVIAL'
}
// Fallback: unknown type treated as moderate
return 'MODERATE'
}
// ─── Tests ─────────────────────────────────────────────────────────────────
describe('Challenge Difficulty Audit', () => {
const templates = buildMergedTemplates()
// Accumulators
const bucketCounts: Record<DifficultyBucket, number> = {
TRIVIAL: 0,
MODERATE: 0,
HARD: 0,
TRICK: 0,
CREATIVE: 0,
}
const typeBreakdown: Record<string, Record<DifficultyBucket, number>> = {}
const typeTotals: Record<string, number> = {}
let totalPrompts = 0
// ─── 1. Classify every prompt ──────────────────────────────────────────
it('classifies all prompts by LLM difficulty', () => {
for (const t of templates) {
typeBreakdown[t.type] = { TRIVIAL: 0, MODERATE: 0, HARD: 0, TRICK: 0, CREATIVE: 0 }
typeTotals[t.type] = t.prompts.length
for (const p of t.prompts) {
totalPrompts++
const bucket = classifyPrompt(t.type, t.scoring, p)
bucketCounts[bucket]++
typeBreakdown[t.type][bucket]++
}
}
expect(totalPrompts).toBeGreaterThan(800)
console.log(`\n Total prompts classified: ${totalPrompts}`)
})
// ─── 2. Verify distribution is reasonable ──────────────────────────────
it('TRIVIAL prompts make up the majority of factual prompts', () => {
const factualTotal = bucketCounts.TRIVIAL + bucketCounts.MODERATE + bucketCounts.HARD + bucketCounts.TRICK
const trivialPct = (bucketCounts.TRIVIAL / factualTotal) * 100
console.log(`\n Factual total: ${factualTotal}`)
console.log(` TRIVIAL: ${bucketCounts.TRIVIAL} (${trivialPct.toFixed(1)}%)`)
console.log(` Hypothesis: 80%+ are TRIVIAL → actual: ${trivialPct.toFixed(1)}%`)
// The hypothesis: most prompts are trivial for an LLM
// We assert at least 30% are trivial (conservative lower bound)
expect(trivialPct).toBeGreaterThan(30)
})
// ─── 3. Every bucket has prompts ───────────────────────────────────────
it('every non-CREATIVE difficulty bucket has at least 1 prompt', () => {
expect(bucketCounts.TRIVIAL).toBeGreaterThan(0)
expect(bucketCounts.MODERATE).toBeGreaterThan(0)
expect(bucketCounts.HARD).toBeGreaterThan(0)
expect(bucketCounts.TRICK).toBeGreaterThan(0)
})
// ─── 4. Type coverage: every type is classified ────────────────────────
it('every challenge type has a difficulty classification', () => {
for (const t of templates) {
const breakdown = typeBreakdown[t.type]
const sum = Object.values(breakdown).reduce((a, b) => a + b, 0)
expect(sum).toBe(t.prompts.length)
}
})
// ─── 5. Trap Card prompts are all TRICK ────────────────────────────────
it('all trap_card prompts are classified as TRICK', () => {
const trapTemplate = templates.find(t => t.type === 'trap_card')
expect(trapTemplate).toBeDefined()
if (trapTemplate) {
expect(typeBreakdown['trap_card'].TRICK).toBe(trapTemplate.prompts.length)
}
})
// ─── 6. Riddle and magic_duel prompts are HARD ─────────────────────────
it('riddle and magic_duel prompts are classified as HARD', () => {
const riddleTemplate = templates.find(t => t.type === 'riddle')
const magicTemplate = templates.find(t => t.type === 'magic_duel')
if (riddleTemplate) {
expect(typeBreakdown['riddle'].HARD).toBe(riddleTemplate.prompts.length)
}
if (magicTemplate) {
expect(typeBreakdown['magic_duel'].HARD).toBe(magicTemplate.prompts.length)
}
})
// ─── 7. Log distribution matrix + write markdown ──────────────────────
it('logs distribution matrix and writes audit markdown', () => {
const buckets: DifficultyBucket[] = ['TRIVIAL', 'MODERATE', 'HARD', 'TRICK', 'CREATIVE']
const factualTotal = bucketCounts.TRIVIAL + bucketCounts.MODERATE + bucketCounts.HARD + bucketCounts.TRICK
// ── Console output ──
console.log(`\n${'═'.repeat(80)}`)
console.log(` CHALLENGE DIFFICULTY AUDIT — DISTRIBUTION MATRIX`)
console.log(`${'═'.repeat(80)}`)
console.log(` Total prompts: ${totalPrompts}`)
console.log(` Factual: ${factualTotal} | Creative: ${bucketCounts.CREATIVE}`)
console.log(`${'─'.repeat(80)}`)
// Header
const header = ' Type'.padEnd(24) + buckets.map(b => b.padStart(10)).join('') + ' TOTAL'
console.log(header)
console.log(`${'─'.repeat(80)}`)
// Per-type rows
const sortedTypes = Object.keys(typeBreakdown).sort()
for (const type of sortedTypes) {
const row = typeBreakdown[type]
const total = typeTotals[type]
const line = ` ${type}`.padEnd(24) + buckets.map(b => String(row[b]).padStart(10)).join('') + String(total).padStart(10)
console.log(line)
}
console.log(`${'─'.repeat(80)}`)
const totalLine = ' TOTAL'.padEnd(24) + buckets.map(b => String(bucketCounts[b]).padStart(10)).join('') + String(totalPrompts).padStart(10)
console.log(totalLine)
// Percentages
const pctLine = ' % of factual'.padEnd(24) + buckets.map(b => {
if (b === 'CREATIVE') return 'N/A'.padStart(10)
return `${((bucketCounts[b] / factualTotal) * 100).toFixed(1)}%`.padStart(10)
}).join('') + ''
console.log(pctLine)
console.log(`${'═'.repeat(80)}`)
// Hypothesis check
const trivialPct = (bucketCounts.TRIVIAL / factualTotal) * 100
console.log(`\n HYPOTHESIS: 80%+ are TRIVIAL`)
console.log(` RESULT: ${trivialPct.toFixed(1)}% are TRIVIAL`)
if (trivialPct >= 80) {
console.log(` VERDICT: CONFIRMED — ${trivialPct.toFixed(1)}% >= 80%`)
} else {
console.log(` VERDICT: REJECTED — ${trivialPct.toFixed(1)}% < 80%`)
console.log(` (Distribution is more varied than hypothesized)`)
}
console.log('')
// ── Write markdown file ──
const __filename = fileURLToPath(import.meta.url)
const projectRoot = resolve(dirname(__filename), '..', '..', '..')
const loopDir = resolve(projectRoot, 'loop')
if (!existsSync(loopDir)) mkdirSync(loopDir, { recursive: true })
const mdPath = resolve(loopDir, 'challenge-difficulty-audit.md')
let md = `# Challenge Difficulty Audit\n\n`
md += `Generated: ${new Date().toISOString().slice(0, 10)}\n\n`
md += `## Summary\n\n`
md += `- **Total prompts**: ${totalPrompts}\n`
md += `- **Factual prompts**: ${factualTotal}\n`
md += `- **Creative prompts**: ${bucketCounts.CREATIVE}\n\n`
md += `## Difficulty Buckets\n\n`
md += `| Bucket | Count | % of Factual | Description |\n`
md += `|--------|------:|-------------:|-------------|\n`
md += `| TRIVIAL | ${bucketCounts.TRIVIAL} | ${((bucketCounts.TRIVIAL / factualTotal) * 100).toFixed(1)}% | 99%+ LLMs correct (basic facts, simple arithmetic) |\n`
md += `| MODERATE | ${bucketCounts.MODERATE} | ${((bucketCounts.MODERATE / factualTotal) * 100).toFixed(1)}% | 10-30% fail rate (niche knowledge, multi-step) |\n`
md += `| HARD | ${bucketCounts.HARD} | ${((bucketCounts.HARD / factualTotal) * 100).toFixed(1)}% | 30%+ fail rate (lateral thinking, adversarial) |\n`
md += `| TRICK | ${bucketCounts.TRICK} | ${((bucketCounts.TRICK / factualTotal) * 100).toFixed(1)}% | Instruction-following tests (prompt injection) |\n`
md += `| CREATIVE | ${bucketCounts.CREATIVE} | N/A | Heuristic scoring, not factual |\n\n`
md += `## Hypothesis\n\n`
md += `> 80%+ of factual prompts are TRIVIAL for an LLM.\n\n`
if (trivialPct >= 80) {
md += `**CONFIRMED**: ${trivialPct.toFixed(1)}% of factual prompts are TRIVIAL.\n\n`
} else {
md += `**REJECTED**: Only ${trivialPct.toFixed(1)}% of factual prompts are TRIVIAL.\n`
md += `The distribution is more varied than hypothesized.\n\n`
}
md += `## Per-Type Breakdown\n\n`
md += `| Type | TRIVIAL | MODERATE | HARD | TRICK | Total |\n`
md += `|------|--------:|---------:|-----:|------:|------:|\n`
for (const type of sortedTypes) {
const row = typeBreakdown[type]
const total = typeTotals[type]
md += `| ${type} | ${row.TRIVIAL} | ${row.MODERATE} | ${row.HARD} | ${row.TRICK} | ${total} |\n`
}
md += `| **TOTAL** | **${bucketCounts.TRIVIAL}** | **${bucketCounts.MODERATE}** | **${bucketCounts.HARD}** | **${bucketCounts.TRICK}** | **${totalPrompts}** |\n\n`
md += `## Classification Methodology\n\n`
md += `Heuristic classification based on challenge type characteristics:\n\n`
md += `- **TRIVIAL types**: speed_blitz, sports_showdown, vehicle_mayhem, animal_kingdom, nature_clash\n`
md += ` - Basic facts, common knowledge, well-known answers\n`
md += `- **MODERATE types**: math_blitz, hallucination_check (false answers), hack_battle, code_golf, creative_writing, meme_war, roast_battle, wrestling_match\n`
md += ` - Specialized knowledge, multi-step reasoning, niche trivia\n`
md += `- **HARD types**: riddle, magic_duel, hallucination_check (true/counter-intuitive answers)\n`
md += ` - Lateral thinking, misdirection, counter-intuitive facts\n`
md += `- **TRICK types**: trap_card\n`
md += ` - Prompt injection resistance, instruction-following under adversarial pressure\n`
md += `- Per-prompt \`difficulty\` tags (easy/medium/hard) further refine classification within moderate types\n`
writeFileSync(mdPath, md)
console.log(` Wrote audit results to: ${mdPath}`)
// Assertion: file was written
expect(existsSync(mdPath)).toBe(true)
})
})