fix(aiui): content cards carried the PREVIOUS item's description
Operator-reported via a Bitcoin-films transcript: "Banking on Bitcoin" was captioned with *The Rise and Rise of Bitcoin*'s description, "Cryptopia" with *The Bitcoin Standard*'s, and the section header "Documentaries:" bled into the first card of each group. Read as the model talking nonsense; the model's prose was correct throughout and only the pairing was wrong. Several patterns anchor with `(?:^|\n)` so they fire only at a line start. That makes m.index point at the NEWLINE — one character before the line the match is really on — so extractDescriptionForTag's window, which walks back from `matchIndex - 1`, landed on the PREVIOUS line. The description became "previous line + this item's own text". Normalised inside the helper rather than at each of its nine call sites, so a pattern that gains a `(?:^|\n)` anchor later cannot silently reintroduce it. Fault-injected to prove the tests are not vacuous: with the fix removed, two fail with exactly the reported strings — 'Documentaries: – Early documentary fo…' and 'The Rise and Rise of Bitcoin – Early …'. 26/26 with it restored. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
4891e1babc
commit
086d381c4f
@@ -123,3 +123,44 @@ And a book:
|
|||||||
expect(titles.length).toBeGreaterThan(0)
|
expect(titles.length).toBeGreaterThan(0)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
describe('description pairing (operator-reported: cards carried the WRONG description)', () => {
|
||||||
|
// Verbatim shape of the Bitcoin-films answer that produced mis-paired cards.
|
||||||
|
// The model's prose was correct; the parser paired each title with the
|
||||||
|
// PREVIOUS line because `(?:^|\n)`-anchored patterns report m.index at the
|
||||||
|
// newline, shifting the line window back by one line.
|
||||||
|
const TRANSCRIPT = [
|
||||||
|
'Here are some films about Bitcoin:',
|
||||||
|
'',
|
||||||
|
'Documentaries:',
|
||||||
|
'',
|
||||||
|
'- The Rise and Rise of Bitcoin (2014) – Early documentary following Bitcoin\'s emergence and community.',
|
||||||
|
'- Banking on Bitcoin (2016) – Explores Bitcoin\'s origins and its potential to disrupt finance.',
|
||||||
|
'- Cryptopia (2020) – Examines both the promise and pitfalls of crypto.',
|
||||||
|
].join('\n')
|
||||||
|
|
||||||
|
it('does not caption a title with the previous item\'s description', () => {
|
||||||
|
const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series')
|
||||||
|
const banking = series.find(s => /Banking on Bitcoin/i.test(s.title))
|
||||||
|
if (banking?.synopsis) {
|
||||||
|
expect(banking.synopsis).not.toMatch(/Early documentary/i)
|
||||||
|
expect(banking.synopsis).toMatch(/Explores Bitcoin/i)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
it('does not bleed a section header into the first card of a group', () => {
|
||||||
|
const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series')
|
||||||
|
for (const s of series) {
|
||||||
|
expect(s.synopsis ?? '').not.toMatch(/Documentaries:/i)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
it('keeps each item paired with its OWN description', () => {
|
||||||
|
const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series')
|
||||||
|
const rise = series.find(s => /Rise and Rise/i.test(s.title))
|
||||||
|
if (rise?.synopsis) {
|
||||||
|
expect(rise.synopsis).toMatch(/Early documentary/i)
|
||||||
|
expect(rise.synopsis).not.toMatch(/Explores Bitcoin/i)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|||||||
@@ -70,6 +70,24 @@ function extractFirstImageFromText(text: string): string | undefined {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function extractDescriptionForTag(text: string, matchIndex: number, matchLength: number): string {
|
function extractDescriptionForTag(text: string, matchIndex: number, matchLength: number): string {
|
||||||
|
// Several patterns anchor with `(?:^|\n)` so they only fire at a line start.
|
||||||
|
// That makes `m.index` point at the NEWLINE, one character before the line
|
||||||
|
// the match is actually on — and the line window below then walks back from
|
||||||
|
// `matchIndex - 1` and lands on the PREVIOUS line. The description became
|
||||||
|
// "previous line + this item's own text".
|
||||||
|
//
|
||||||
|
// Operator-visible result: "Banking on Bitcoin" captioned with *The Rise and
|
||||||
|
// Rise of Bitcoin*'s description, and the section header "Documentaries:"
|
||||||
|
// bleeding into the first card of each group. The model's prose was correct
|
||||||
|
// throughout; only this pairing was wrong.
|
||||||
|
//
|
||||||
|
// Normalised here rather than at each of the nine call sites, so a pattern
|
||||||
|
// that gains a `(?:^|\n)` anchor later cannot silently reintroduce it.
|
||||||
|
while (matchLength > 0 && (text[matchIndex] === '\n' || text[matchIndex] === '\r')) {
|
||||||
|
matchIndex += 1
|
||||||
|
matchLength -= 1
|
||||||
|
}
|
||||||
|
|
||||||
const prevNewline = text.lastIndexOf('\n', matchIndex - 1)
|
const prevNewline = text.lastIndexOf('\n', matchIndex - 1)
|
||||||
const lineStart = prevNewline === -1 ? 0 : prevNewline + 1
|
const lineStart = prevNewline === -1 ? 0 : prevNewline + 1
|
||||||
const nextNewline = text.indexOf('\n', matchIndex + matchLength)
|
const nextNewline = text.indexOf('\n', matchIndex + matchLength)
|
||||||
|
|||||||
Reference in New Issue
Block a user