From 086d381c4f373116c0d0902b23ebcdea1d61a037 Mon Sep 17 00:00:00 2001 From: archipelago Date: Thu, 6 Aug 2026 21:08:11 -0400 Subject: [PATCH] fix(aiui): content cards carried the PREVIOUS item's description MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Operator-reported via a Bitcoin-films transcript: "Banking on Bitcoin" was captioned with *The Rise and Rise of Bitcoin*'s description, "Cryptopia" with *The Bitcoin Standard*'s, and the section header "Documentaries:" bled into the first card of each group. Read as the model talking nonsense; the model's prose was correct throughout and only the pairing was wrong. Several patterns anchor with `(?:^|\n)` so they fire only at a line start. That makes m.index point at the NEWLINE — one character before the line the match is really on — so extractDescriptionForTag's window, which walks back from `matchIndex - 1`, landed on the PREVIOUS line. The description became "previous line + this item's own text". Normalised inside the helper rather than at each of its nine call sites, so a pattern that gains a `(?:^|\n)` anchor later cannot silently reintroduce it. Fault-injected to prove the tests are not vacuous: with the fix removed, two fail with exactly the reported strings — 'Documentaries: – Early documentary fo…' and 'The Rise and Rise of Bitcoin – Early …'. 26/26 with it restored. Co-Authored-By: Claude Opus 5 (1M context) --- .../__tests__/contentExtraction.test.ts | 41 +++++++++++++++++++ .../app/src/composables/contentExtraction.ts | 18 ++++++++ 2 files changed, 59 insertions(+) diff --git a/aiui/packages/app/src/composables/__tests__/contentExtraction.test.ts b/aiui/packages/app/src/composables/__tests__/contentExtraction.test.ts index c7dfa040..84fabbe8 100644 --- a/aiui/packages/app/src/composables/__tests__/contentExtraction.test.ts +++ b/aiui/packages/app/src/composables/__tests__/contentExtraction.test.ts @@ -123,3 +123,44 @@ And a book: expect(titles.length).toBeGreaterThan(0) }) }) + +describe('description pairing (operator-reported: cards carried the WRONG description)', () => { + // Verbatim shape of the Bitcoin-films answer that produced mis-paired cards. + // The model's prose was correct; the parser paired each title with the + // PREVIOUS line because `(?:^|\n)`-anchored patterns report m.index at the + // newline, shifting the line window back by one line. + const TRANSCRIPT = [ + 'Here are some films about Bitcoin:', + '', + 'Documentaries:', + '', + '- The Rise and Rise of Bitcoin (2014) – Early documentary following Bitcoin\'s emergence and community.', + '- Banking on Bitcoin (2016) – Explores Bitcoin\'s origins and its potential to disrupt finance.', + '- Cryptopia (2020) – Examines both the promise and pitfalls of crypto.', + ].join('\n') + + it('does not caption a title with the previous item\'s description', () => { + const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series') + const banking = series.find(s => /Banking on Bitcoin/i.test(s.title)) + if (banking?.synopsis) { + expect(banking.synopsis).not.toMatch(/Early documentary/i) + expect(banking.synopsis).toMatch(/Explores Bitcoin/i) + } + }) + + it('does not bleed a section header into the first card of a group', () => { + const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series') + for (const s of series) { + expect(s.synopsis ?? '').not.toMatch(/Documentaries:/i) + } + }) + + it('keeps each item paired with its OWN description', () => { + const series = extractAllTVSeries(TRANSCRIPT, 'recommend me some bitcoin series') + const rise = series.find(s => /Rise and Rise/i.test(s.title)) + if (rise?.synopsis) { + expect(rise.synopsis).toMatch(/Early documentary/i) + expect(rise.synopsis).not.toMatch(/Explores Bitcoin/i) + } + }) +}) diff --git a/aiui/packages/app/src/composables/contentExtraction.ts b/aiui/packages/app/src/composables/contentExtraction.ts index d7dfd111..638fc783 100644 --- a/aiui/packages/app/src/composables/contentExtraction.ts +++ b/aiui/packages/app/src/composables/contentExtraction.ts @@ -70,6 +70,24 @@ function extractFirstImageFromText(text: string): string | undefined { } function extractDescriptionForTag(text: string, matchIndex: number, matchLength: number): string { + // Several patterns anchor with `(?:^|\n)` so they only fire at a line start. + // That makes `m.index` point at the NEWLINE, one character before the line + // the match is actually on — and the line window below then walks back from + // `matchIndex - 1` and lands on the PREVIOUS line. The description became + // "previous line + this item's own text". + // + // Operator-visible result: "Banking on Bitcoin" captioned with *The Rise and + // Rise of Bitcoin*'s description, and the section header "Documentaries:" + // bleeding into the first card of each group. The model's prose was correct + // throughout; only this pairing was wrong. + // + // Normalised here rather than at each of the nine call sites, so a pattern + // that gains a `(?:^|\n)` anchor later cannot silently reintroduce it. + while (matchLength > 0 && (text[matchIndex] === '\n' || text[matchIndex] === '\r')) { + matchIndex += 1 + matchLength -= 1 + } + const prevNewline = text.lastIndexOf('\n', matchIndex - 1) const lineStart = prevNewline === -1 ? 0 : prevNewline + 1 const nextNewline = text.indexOf('\n', matchIndex + matchLength)