feat(content): comprehensive extraction hardening, code mode UI, overnight plan

- Relax isBookLikeResponse threshold (>=2 to >=1)
- Widen book patterns: inline prose, verb-preceded, numbered bold, em-dash
- Remove overly aggressive film/song gating on book extraction
- Allow songs to coexist with film/book tags (explicit tags always returned)
- Strengthen TV patterns: seasons, created by, standalone "tv" query match
- Fix TV_EXT_RE to handle both Title|Year|Creator and Title|Creator|Year
- Widen place patterns: type word search in descriptions, bold fallback
- Add "pizza" to isPlaceQuery and preferredFirstTab
- Add Code tab: isCodeQuery, isCodeLikeResponse, extractCodeBlocks, wiring
- Fix image threshold: single image with meaningful alt text shown
- Refactor filterTabsByContext: specialized paths now append remaining content
- Add code mode UI: orange input styling, design system selection, file selection
- DesignSystemGrid: selection toggle only on checkmark, card click opens detail
- Add 64-test extraction quality test suite
- Update overnight plan.md and prompt.md for hardening run

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Dorian
2026-03-04 10:54:03 +00:00
co-authored by Claude Opus 4.6
parent 75e9274582
commit f0eb4383bd
13 changed files with 1049 additions and 260 deletions
+127 -20
View File
@@ -16,7 +16,7 @@ export const PODCAST_TAG_RE = /\[\[podcast:(p?\d+)\]\]/gi
export const PODCAST_EXT_RE = /\[\[podcast_ext:([^|]+)\|([^|]+)(?:\|(\d{4}))?\]\]/gi
export const BOOK_TAG_RE = /\[\[book:(b?\d+)\]\]/gi
export const BOOK_EXT_RE = /\[\[book_ext:([^|]+)\|([^|]+)(?:\|(\d{4}))?\]\]/gi
export const TV_EXT_RE = /\[\[tv_ext:([^|]+)\|([^|]*)(?:\|(\d{4}))?\]\]/gi
export const TV_EXT_RE = /\[\[tv_ext:([^|]+)\|([^|\]]+)(?:\|([^|\]]+))?\]\]/gi
export const PLACE_EXT_RE = /\[\[place_ext:([^|]+)\|([^|]*)(?:\|([^|]*))?(?:\|([^|]*))?(?:\|([^|]*))?(?:\|([^|]*))?\]\]/gi
export const MARKDOWN_LINK_RE = /\[([^\]]+)\]\((https?:\/\/[^)\s]+)\)/g
const SAFE_URL_SCHEME = /^https?:\/\//i
@@ -602,12 +602,25 @@ function extractSongsFromPatterns(text: string): Song[] {
export function extractAllSongs(text: string, userQuery = ''): Song[] {
const librarySongs = resolveSongs(extractSongIds(text))
const externalSongs = extractExternalSongs(text)
if (librarySongs.length > 0 || externalSongs.length > 0) {
return [...librarySongs, ...externalSongs]
// Explicit song tags and library matches always returned
const explicitSongs = [...librarySongs, ...externalSongs]
// Skip pattern-based fallback when conflicting content tags are present
const hasConflictingTags = extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text) ||
extractPodcastIds(text).length > 0 || /\[\[podcast_ext:/.test(text) ||
/\[\[tv_ext:/.test(text) || /\[\[book_ext:/.test(text)
if (explicitSongs.length > 0) {
if (hasConflictingTags) return explicitSongs
// Also grab pattern matches alongside explicit songs
const libMatches = extractSongsFromLibraryMatch(text)
const patternMatches = extractSongsFromPatterns(text)
const existingKeys = new Set(explicitSongs.map((s) => `${s.title.toLowerCase()}|${s.artist.toLowerCase()}`))
const extras = [...libMatches, ...patternMatches].filter(
(p) => !existingKeys.has(`${p.title.toLowerCase()}|${p.artist.toLowerCase()}`)
)
return [...explicitSongs, ...extras]
}
if (extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text)) return []
if (extractPodcastIds(text).length > 0 || /\[\[podcast_ext:/.test(text)) return []
if (/\[\[tv_ext:/.test(text) || /\[\[book_ext:/.test(text)) return []
if (hasConflictingTags) return []
if (isNewsLikeResponse(text)) return []
const q = userQuery.toLowerCase()
if (q && !isMusicQuery(q)) return []
@@ -723,9 +736,20 @@ function extractBooksFromPatterns(text: string): Book[] {
const seen = new Set<string>()
const patterns: { re: RegExp; titleIdx: number; authorIdx: number }[] = [
{ re: /["""]([^"""]{2,80})["""]\s+by\s+([A-Z][^,\n.]{1,50}?)(?:\s*[,()\n.]|$)/gi, titleIdx: 1, authorIdx: 2 },
{ re: /\*\*([^*]{2,80})\*\*\s+(?:by|—|)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s*[,()*\n]|$)/g, titleIdx: 1, authorIdx: 2 },
{ re: /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n\-–—]{2,80}?)\*{0,2}\s+(?:by|—|)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s*[,()*\n]|$)/gm, titleIdx: 1, authorIdx: 2 },
// "Title" by Author or "Title" by Author (straight and smart quotes)
{ re: /[""\u201C\u201D]([^"""\u201C\u201D]{2,80})[""\u201C\u201D]\s+by\s+([A-Z][^,\n]{1,50}?)(?:\s+(?:is|was|has|—||-)|[,()\n.]|$)/gi, titleIdx: 1, authorIdx: 2 },
// **Title** by/—/ Author
{ re: /\*\*([^*]{2,80})\*\*\s+(?:by|—|)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s+(?:is|was|has|—||-)|[,()*\n.]|$)/g, titleIdx: 1, authorIdx: 2 },
// List item: "1. Title by Author" or "- Title by Author"
{ re: /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n\-–—]{2,80}?)\*{0,2}\s+(?:by|—|)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s+(?:is|was|has|—||-)|[,()*\n.]|$)/gm, titleIdx: 1, authorIdx: 2 },
// Numbered list: "1. **Title** by Author — Description"
{ re: /(?:^|\n)\s*\d+\.\s*\*\*([^*]{2,80})\*\*\s+by\s+([A-Z][^,\n(—–-]{1,50}?)(?:\s*[,()\n—–-]|$)/gm, titleIdx: 1, authorIdx: 2 },
// Em-dash list: "- Title — Author (year)" or "- Title — Author"
{ re: /(?:^|\n)\s*[-•]\s+([^—–\n]{2,80}?)\s+[—–]\s+([A-Z][^,\n(]{1,50}?)(?:\s*\(\d{4}\))?(?:\s*[,\n]|$)/gm, titleIdx: 1, authorIdx: 2 },
// Inline prose at line/sentence start: "Title Words by Author"
{ re: /(?:^|\n|[.!?]\s+)([A-Z][a-z]+(?:\s+[A-Z][a-z]*){1,8})\s+by\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){0,4})(?:\s+[a-z]|[,.()\n—–-]|$)/gm, titleIdx: 1, authorIdx: 2 },
// After recommendation verbs: "enjoy Title by Author"
{ re: /(?:read|enjoy|recommend|check out|try|start with|pick up)\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){1,8})\s+by\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){0,4})(?:\s+[a-z]|[,.()\n—–-]|$)/gi, titleIdx: 1, authorIdx: 2 },
]
for (const { re, titleIdx, authorIdx } of patterns) {
@@ -764,10 +788,6 @@ export function extractAllBooks(text: string, userQuery: string): Book[] {
const externalBooks = extractExternalBooks(text)
if (externalBooks.length > 0) return externalBooks
if (!isBookQuery(userQuery) && !isBookLikeResponse(text)) return []
if (!isBookQuery(userQuery)) {
if (extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text)) return []
if (extractSongIds(text).length > 0 || /\[\[song_ext:/.test(text)) return []
}
if (isNewsLikeResponse(text)) return []
const cleanText = text.replace(/\n---\n\s*(?:Sources|References|Links):?\s*\n[\s\S]*$/i, '')
return extractBooksFromPatterns(cleanText)
@@ -782,8 +802,13 @@ function extractExternalTVSeries(text: string): TVSeries[] {
const re = new RegExp(TV_EXT_RE.source, 'gi')
while ((match = re.exec(text)) !== null) {
const title = match[1].trim()
const creator = match[2].trim()
const year = match[3] ? parseInt(match[3], 10) : undefined
const field2 = match[2].trim()
const field3 = match[3]?.trim()
// Handle both Title|Year|Creator and Title|Creator|Year formats
const isField2Year = /^\d{4}$/.test(field2)
const year = isField2Year ? parseInt(field2, 10)
: field3 && /^\d{4}$/.test(field3) ? parseInt(field3, 10) : undefined
const creator = isField2Year ? (field3 || undefined) : (field2 || undefined)
const key = title.toLowerCase()
if (seen.has(key)) continue
seen.add(key)
@@ -801,8 +826,18 @@ function extractExternalTVSeries(text: string): TVSeries[] {
}
function isTVLikeResponse(text: string): boolean {
return /\b(season|episodes?|showrunner|streaming|renewed|cancelled|premiere|network|HBO|Netflix|AMC|FX|Apple TV|Disney\+)\b/i.test(text) &&
(text.match(/\bseason\b/gi)?.length ?? 0) >= 2
const hasKeyword = /\b(season|episodes?|showrunner|streaming|renewed|cancelled|premiere|network|HBO|Netflix|AMC|FX|Apple TV|Disney\+|created by)\b/i.test(text)
if (!hasKeyword) return false
// Single "season" mention is sufficient
if ((text.match(/\bseason\b/gi)?.length ?? 0) >= 1) return true
// Multiple distinct TV signals without requiring "season"
const tvSignals = [
/\bepisodes?\b/i, /\bshowrunner\b/i, /\bstreaming\b/i, /\brenewed\b/i,
/\bcancelled\b/i, /\bpremiere\b/i, /\bnetwork\b/i,
/\b(HBO|Netflix|AMC|FX|Apple TV|Disney\+|Hulu|Amazon Prime)\b/i,
/\bseries\b/i, /\bpilot\b/i, /\bminiseries\b/i, /\bcreated by\b/i,
]
return tvSignals.filter(re => re.test(text)).length >= 2
}
function extractTVSeriesFromPatterns(text: string): TVSeries[] {
@@ -812,6 +847,10 @@ function extractTVSeriesFromPatterns(text: string): TVSeries[] {
/"([^"]{2,60})"\s*[-–—]\s*(?:a |an )?(?:series|show|tv)/gi,
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*(?:a |an )?(?:\w+ )?(?:series|show|drama|comedy|thriller|animated)/gi,
/(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n]{2,60}?)\*{0,2}\s*\((\d{4})(?:[-]\d{0,4})?(?:,\s*\d+ seasons?)?\)/gm,
// Bold title with seasons info: **Title** — 5 seasons
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*\d+\s*seasons?/gi,
// "Title" — creator (in TV context)
/"([^"]{2,60})"\s*[-–—]\s*([A-Z][^,\n.]{1,50}?)(?:\s*[,()\n]|$)/gi,
]
for (const re of patterns) {
let m: RegExpExecArray | null
@@ -918,6 +957,8 @@ export function extractAllImages(text: string, userQuery: string): ImageItem[] {
const images = extractImages(text)
if (images.length === 0) return []
if (isImageQuery(userQuery) || images.length >= 2) return images
// Show single image if it has meaningful alt text (intentional image context)
if (images.length === 1 && images[0].alt && images[0].alt.length > 2) return images
return []
}
@@ -957,9 +998,16 @@ function extractPlacesFromPatterns(text: string): Place[] {
const places: { name: string; cuisine?: string; city?: string; rating?: number; priceLevel?: number; desc: string; pos: number }[] = []
const seen = new Set<string>()
const placeTypeWords = 'restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|eatery|spot|joint|trattoria|taqueria|brasserie|cantina|chophouse|creamery|diner|tavern'
const patterns: RegExp[] = [
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*(?:a |an )?(?:(\w[\w\s]{1,30}?)\s+)?(?:restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|eatery|spot|joint)/gi,
/(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n]{2,60}?)\*{0,2}\s*[-–—(]\s*(?:(\w[\w\s&]{1,30}?)\s+)?(?:restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|cuisine|food|dining)/gim,
// **Name** — ... type word (within 80 chars of dash)
new RegExp(`\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*(?:a |an )?(?:(\\w[\\w\\s]{1,30}?)\\s+)?(?:${placeTypeWords})`, 'gi'),
// List item with type word
new RegExp(`(?:^|\\n)\\s*(?:\\d+\\.\\s*|[-•]\\s*)\\*{0,2}([^*\\n]{2,60}?)\\*{0,2}\\s*[-–—(]\\s*(?:(\\w[\\w\\s&]{1,30}?)\\s+)?(?:${placeTypeWords}|cuisine|food|dining)`, 'gim'),
// **Name** — description containing a place type word anywhere in the next 120 chars
new RegExp(`\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*([^\\n]{3,120}?\\b(?:${placeTypeWords})\\b[^\\n]{0,40})`, 'gi'),
// List: 1. **Name** — description with place type word
new RegExp(`(?:^|\\n)\\s*(?:\\d+\\.\\s*|[-•]\\s*)\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*([^\\n]{3,120}?\\b(?:${placeTypeWords})\\b[^\\n]{0,40})`, 'gim'),
]
for (const re of patterns) {
@@ -996,12 +1044,71 @@ function extractPlacesFromPatterns(text: string): Place[] {
}))
}
function extractPlacesFromBoldPatterns(text: string): Place[] {
const places: { name: string; desc: string; pos: number }[] = []
const seen = new Set<string>()
// Match any bold title followed by dash/colon and description (for place-query context)
const re = /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*\*([^*]{2,60})\*\*\s*[-–—:]\s*([^\n]{5,200})/gm
let m: RegExpExecArray | null
while ((m = re.exec(text)) !== null) {
const name = m[1].trim()
if (name.length < 2) continue
if (/\[\[(film|song|podcast|book|tv|place)(_ext)?:/.test(name)) continue
const key = name.toLowerCase()
if (seen.has(key)) continue
seen.add(key)
places.push({ name, desc: m[2].trim(), pos: m.index })
}
return places
.sort((a, b) => a.pos - b.pos)
.map(({ name, desc }) => ({
id: `ext-place-${name.toLowerCase().replace(/\W/g, '-')}`,
name,
description: desc,
sources: [],
}))
}
export function extractAllPlaces(text: string, userQuery: string): Place[] {
const external = extractExternalPlaces(text)
if (external.length > 0) return external
if (!isPlaceQuery(userQuery) && !isPlaceLikeResponse(text)) return []
if (isNewsLikeResponse(text)) return []
return extractPlacesFromPatterns(text)
const fromPatterns = extractPlacesFromPatterns(text)
// For place queries, supplement with bold-title patterns to catch items without type words
if (isPlaceQuery(userQuery)) {
const fromBold = extractPlacesFromBoldPatterns(text)
const seenNames = new Set(fromPatterns.map(p => p.name.toLowerCase()))
const extras = fromBold.filter(p => !seenNames.has(p.name.toLowerCase()))
const combined = [...fromPatterns, ...extras]
return combined.length > 0 ? combined : []
}
return fromPatterns
}
// ─── Code block extraction ─────────────────────────────────────────
export interface CodeBlock {
language: string
code: string
label?: string
}
export function extractCodeBlocks(text: string): CodeBlock[] {
const blocks: CodeBlock[] = []
const re = /```(\w*)\n([\s\S]*?)```/g
let m: RegExpExecArray | null
while ((m = re.exec(text)) !== null) {
const language = m[1] || 'text'
const code = m[2].trimEnd()
if (code.length < 3) continue
// Try to find a label from the line above the code block
const before = text.slice(Math.max(0, m.index - 200), m.index)
const labelMatch = /(?:^|\n)\s*(?:\*\*([^*]{2,80})\*\*|#+\s+(.{2,80}))\s*\n?\s*$/.exec(before)
const label = labelMatch?.[1] || labelMatch?.[2] || undefined
blocks.push({ language, code, label })
}
return blocks
}
// ─── Tag stripping ────────────────────────────────────────────────