feat(content): comprehensive extraction hardening, code mode UI, overnight plan
- Relax isBookLikeResponse threshold (>=2 to >=1) - Widen book patterns: inline prose, verb-preceded, numbered bold, em-dash - Remove overly aggressive film/song gating on book extraction - Allow songs to coexist with film/book tags (explicit tags always returned) - Strengthen TV patterns: seasons, created by, standalone "tv" query match - Fix TV_EXT_RE to handle both Title|Year|Creator and Title|Creator|Year - Widen place patterns: type word search in descriptions, bold fallback - Add "pizza" to isPlaceQuery and preferredFirstTab - Add Code tab: isCodeQuery, isCodeLikeResponse, extractCodeBlocks, wiring - Fix image threshold: single image with meaningful alt text shown - Refactor filterTabsByContext: specialized paths now append remaining content - Add code mode UI: orange input styling, design system selection, file selection - DesignSystemGrid: selection toggle only on checkmark, card click opens detail - Add 64-test extraction quality test suite - Update overnight plan.md and prompt.md for hardening run Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
75e9274582
commit
f0eb4383bd
@@ -16,7 +16,7 @@ export const PODCAST_TAG_RE = /\[\[podcast:(p?\d+)\]\]/gi
|
||||
export const PODCAST_EXT_RE = /\[\[podcast_ext:([^|]+)\|([^|]+)(?:\|(\d{4}))?\]\]/gi
|
||||
export const BOOK_TAG_RE = /\[\[book:(b?\d+)\]\]/gi
|
||||
export const BOOK_EXT_RE = /\[\[book_ext:([^|]+)\|([^|]+)(?:\|(\d{4}))?\]\]/gi
|
||||
export const TV_EXT_RE = /\[\[tv_ext:([^|]+)\|([^|]*)(?:\|(\d{4}))?\]\]/gi
|
||||
export const TV_EXT_RE = /\[\[tv_ext:([^|]+)\|([^|\]]+)(?:\|([^|\]]+))?\]\]/gi
|
||||
export const PLACE_EXT_RE = /\[\[place_ext:([^|]+)\|([^|]*)(?:\|([^|]*))?(?:\|([^|]*))?(?:\|([^|]*))?(?:\|([^|]*))?\]\]/gi
|
||||
export const MARKDOWN_LINK_RE = /\[([^\]]+)\]\((https?:\/\/[^)\s]+)\)/g
|
||||
const SAFE_URL_SCHEME = /^https?:\/\//i
|
||||
@@ -602,12 +602,25 @@ function extractSongsFromPatterns(text: string): Song[] {
|
||||
export function extractAllSongs(text: string, userQuery = ''): Song[] {
|
||||
const librarySongs = resolveSongs(extractSongIds(text))
|
||||
const externalSongs = extractExternalSongs(text)
|
||||
if (librarySongs.length > 0 || externalSongs.length > 0) {
|
||||
return [...librarySongs, ...externalSongs]
|
||||
// Explicit song tags and library matches always returned
|
||||
const explicitSongs = [...librarySongs, ...externalSongs]
|
||||
|
||||
// Skip pattern-based fallback when conflicting content tags are present
|
||||
const hasConflictingTags = extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text) ||
|
||||
extractPodcastIds(text).length > 0 || /\[\[podcast_ext:/.test(text) ||
|
||||
/\[\[tv_ext:/.test(text) || /\[\[book_ext:/.test(text)
|
||||
if (explicitSongs.length > 0) {
|
||||
if (hasConflictingTags) return explicitSongs
|
||||
// Also grab pattern matches alongside explicit songs
|
||||
const libMatches = extractSongsFromLibraryMatch(text)
|
||||
const patternMatches = extractSongsFromPatterns(text)
|
||||
const existingKeys = new Set(explicitSongs.map((s) => `${s.title.toLowerCase()}|${s.artist.toLowerCase()}`))
|
||||
const extras = [...libMatches, ...patternMatches].filter(
|
||||
(p) => !existingKeys.has(`${p.title.toLowerCase()}|${p.artist.toLowerCase()}`)
|
||||
)
|
||||
return [...explicitSongs, ...extras]
|
||||
}
|
||||
if (extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text)) return []
|
||||
if (extractPodcastIds(text).length > 0 || /\[\[podcast_ext:/.test(text)) return []
|
||||
if (/\[\[tv_ext:/.test(text) || /\[\[book_ext:/.test(text)) return []
|
||||
if (hasConflictingTags) return []
|
||||
if (isNewsLikeResponse(text)) return []
|
||||
const q = userQuery.toLowerCase()
|
||||
if (q && !isMusicQuery(q)) return []
|
||||
@@ -723,9 +736,20 @@ function extractBooksFromPatterns(text: string): Book[] {
|
||||
const seen = new Set<string>()
|
||||
|
||||
const patterns: { re: RegExp; titleIdx: number; authorIdx: number }[] = [
|
||||
{ re: /["""]([^"""]{2,80})["""]\s+by\s+([A-Z][^,\n.]{1,50}?)(?:\s*[,()\n.]|$)/gi, titleIdx: 1, authorIdx: 2 },
|
||||
{ re: /\*\*([^*]{2,80})\*\*\s+(?:by|—|–)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s*[,()*\n]|$)/g, titleIdx: 1, authorIdx: 2 },
|
||||
{ re: /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n\-–—]{2,80}?)\*{0,2}\s+(?:by|—|–)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s*[,()*\n]|$)/gm, titleIdx: 1, authorIdx: 2 },
|
||||
// "Title" by Author or "Title" by Author (straight and smart quotes)
|
||||
{ re: /[""\u201C\u201D]([^"""\u201C\u201D]{2,80})[""\u201C\u201D]\s+by\s+([A-Z][^,\n]{1,50}?)(?:\s+(?:is|was|has|—|–|-)|[,()\n.]|$)/gi, titleIdx: 1, authorIdx: 2 },
|
||||
// **Title** by/—/– Author
|
||||
{ re: /\*\*([^*]{2,80})\*\*\s+(?:by|—|–)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s+(?:is|was|has|—|–|-)|[,()*\n.]|$)/g, titleIdx: 1, authorIdx: 2 },
|
||||
// List item: "1. Title by Author" or "- Title by Author"
|
||||
{ re: /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n\-–—]{2,80}?)\*{0,2}\s+(?:by|—|–)\s+\*?([A-Z][^*\n]{1,50}?)\*?(?:\s+(?:is|was|has|—|–|-)|[,()*\n.]|$)/gm, titleIdx: 1, authorIdx: 2 },
|
||||
// Numbered list: "1. **Title** by Author — Description"
|
||||
{ re: /(?:^|\n)\s*\d+\.\s*\*\*([^*]{2,80})\*\*\s+by\s+([A-Z][^,\n(—–-]{1,50}?)(?:\s*[,()\n—–-]|$)/gm, titleIdx: 1, authorIdx: 2 },
|
||||
// Em-dash list: "- Title — Author (year)" or "- Title — Author"
|
||||
{ re: /(?:^|\n)\s*[-•]\s+([^—–\n]{2,80}?)\s+[—–]\s+([A-Z][^,\n(]{1,50}?)(?:\s*\(\d{4}\))?(?:\s*[,\n]|$)/gm, titleIdx: 1, authorIdx: 2 },
|
||||
// Inline prose at line/sentence start: "Title Words by Author"
|
||||
{ re: /(?:^|\n|[.!?]\s+)([A-Z][a-z]+(?:\s+[A-Z][a-z]*){1,8})\s+by\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){0,4})(?:\s+[a-z]|[,.()\n—–-]|$)/gm, titleIdx: 1, authorIdx: 2 },
|
||||
// After recommendation verbs: "enjoy Title by Author"
|
||||
{ re: /(?:read|enjoy|recommend|check out|try|start with|pick up)\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){1,8})\s+by\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]*){0,4})(?:\s+[a-z]|[,.()\n—–-]|$)/gi, titleIdx: 1, authorIdx: 2 },
|
||||
]
|
||||
|
||||
for (const { re, titleIdx, authorIdx } of patterns) {
|
||||
@@ -764,10 +788,6 @@ export function extractAllBooks(text: string, userQuery: string): Book[] {
|
||||
const externalBooks = extractExternalBooks(text)
|
||||
if (externalBooks.length > 0) return externalBooks
|
||||
if (!isBookQuery(userQuery) && !isBookLikeResponse(text)) return []
|
||||
if (!isBookQuery(userQuery)) {
|
||||
if (extractFilmIds(text).length > 0 || /\[\[film_ext:/.test(text)) return []
|
||||
if (extractSongIds(text).length > 0 || /\[\[song_ext:/.test(text)) return []
|
||||
}
|
||||
if (isNewsLikeResponse(text)) return []
|
||||
const cleanText = text.replace(/\n---\n\s*(?:Sources|References|Links):?\s*\n[\s\S]*$/i, '')
|
||||
return extractBooksFromPatterns(cleanText)
|
||||
@@ -782,8 +802,13 @@ function extractExternalTVSeries(text: string): TVSeries[] {
|
||||
const re = new RegExp(TV_EXT_RE.source, 'gi')
|
||||
while ((match = re.exec(text)) !== null) {
|
||||
const title = match[1].trim()
|
||||
const creator = match[2].trim()
|
||||
const year = match[3] ? parseInt(match[3], 10) : undefined
|
||||
const field2 = match[2].trim()
|
||||
const field3 = match[3]?.trim()
|
||||
// Handle both Title|Year|Creator and Title|Creator|Year formats
|
||||
const isField2Year = /^\d{4}$/.test(field2)
|
||||
const year = isField2Year ? parseInt(field2, 10)
|
||||
: field3 && /^\d{4}$/.test(field3) ? parseInt(field3, 10) : undefined
|
||||
const creator = isField2Year ? (field3 || undefined) : (field2 || undefined)
|
||||
const key = title.toLowerCase()
|
||||
if (seen.has(key)) continue
|
||||
seen.add(key)
|
||||
@@ -801,8 +826,18 @@ function extractExternalTVSeries(text: string): TVSeries[] {
|
||||
}
|
||||
|
||||
function isTVLikeResponse(text: string): boolean {
|
||||
return /\b(season|episodes?|showrunner|streaming|renewed|cancelled|premiere|network|HBO|Netflix|AMC|FX|Apple TV|Disney\+)\b/i.test(text) &&
|
||||
(text.match(/\bseason\b/gi)?.length ?? 0) >= 2
|
||||
const hasKeyword = /\b(season|episodes?|showrunner|streaming|renewed|cancelled|premiere|network|HBO|Netflix|AMC|FX|Apple TV|Disney\+|created by)\b/i.test(text)
|
||||
if (!hasKeyword) return false
|
||||
// Single "season" mention is sufficient
|
||||
if ((text.match(/\bseason\b/gi)?.length ?? 0) >= 1) return true
|
||||
// Multiple distinct TV signals without requiring "season"
|
||||
const tvSignals = [
|
||||
/\bepisodes?\b/i, /\bshowrunner\b/i, /\bstreaming\b/i, /\brenewed\b/i,
|
||||
/\bcancelled\b/i, /\bpremiere\b/i, /\bnetwork\b/i,
|
||||
/\b(HBO|Netflix|AMC|FX|Apple TV|Disney\+|Hulu|Amazon Prime)\b/i,
|
||||
/\bseries\b/i, /\bpilot\b/i, /\bminiseries\b/i, /\bcreated by\b/i,
|
||||
]
|
||||
return tvSignals.filter(re => re.test(text)).length >= 2
|
||||
}
|
||||
|
||||
function extractTVSeriesFromPatterns(text: string): TVSeries[] {
|
||||
@@ -812,6 +847,10 @@ function extractTVSeriesFromPatterns(text: string): TVSeries[] {
|
||||
/"([^"]{2,60})"\s*[-–—]\s*(?:a |an )?(?:series|show|tv)/gi,
|
||||
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*(?:a |an )?(?:\w+ )?(?:series|show|drama|comedy|thriller|animated)/gi,
|
||||
/(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n]{2,60}?)\*{0,2}\s*\((\d{4})(?:[-–]\d{0,4})?(?:,\s*\d+ seasons?)?\)/gm,
|
||||
// Bold title with seasons info: **Title** — 5 seasons
|
||||
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*\d+\s*seasons?/gi,
|
||||
// "Title" — creator (in TV context)
|
||||
/"([^"]{2,60})"\s*[-–—]\s*([A-Z][^,\n.]{1,50}?)(?:\s*[,()\n]|$)/gi,
|
||||
]
|
||||
for (const re of patterns) {
|
||||
let m: RegExpExecArray | null
|
||||
@@ -918,6 +957,8 @@ export function extractAllImages(text: string, userQuery: string): ImageItem[] {
|
||||
const images = extractImages(text)
|
||||
if (images.length === 0) return []
|
||||
if (isImageQuery(userQuery) || images.length >= 2) return images
|
||||
// Show single image if it has meaningful alt text (intentional image context)
|
||||
if (images.length === 1 && images[0].alt && images[0].alt.length > 2) return images
|
||||
return []
|
||||
}
|
||||
|
||||
@@ -957,9 +998,16 @@ function extractPlacesFromPatterns(text: string): Place[] {
|
||||
const places: { name: string; cuisine?: string; city?: string; rating?: number; priceLevel?: number; desc: string; pos: number }[] = []
|
||||
const seen = new Set<string>()
|
||||
|
||||
const placeTypeWords = 'restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|eatery|spot|joint|trattoria|taqueria|brasserie|cantina|chophouse|creamery|diner|tavern'
|
||||
const patterns: RegExp[] = [
|
||||
/\*\*([^*]{2,60})\*\*\s*[-–—:]\s*(?:a |an )?(?:(\w[\w\s]{1,30}?)\s+)?(?:restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|eatery|spot|joint)/gi,
|
||||
/(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*{0,2}([^*\n]{2,60}?)\*{0,2}\s*[-–—(]\s*(?:(\w[\w\s&]{1,30}?)\s+)?(?:restaurant|cafe|bar|bistro|pub|pizzeria|bakery|deli|steakhouse|grill|cuisine|food|dining)/gim,
|
||||
// **Name** — ... type word (within 80 chars of dash)
|
||||
new RegExp(`\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*(?:a |an )?(?:(\\w[\\w\\s]{1,30}?)\\s+)?(?:${placeTypeWords})`, 'gi'),
|
||||
// List item with type word
|
||||
new RegExp(`(?:^|\\n)\\s*(?:\\d+\\.\\s*|[-•]\\s*)\\*{0,2}([^*\\n]{2,60}?)\\*{0,2}\\s*[-–—(]\\s*(?:(\\w[\\w\\s&]{1,30}?)\\s+)?(?:${placeTypeWords}|cuisine|food|dining)`, 'gim'),
|
||||
// **Name** — description containing a place type word anywhere in the next 120 chars
|
||||
new RegExp(`\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*([^\\n]{3,120}?\\b(?:${placeTypeWords})\\b[^\\n]{0,40})`, 'gi'),
|
||||
// List: 1. **Name** — description with place type word
|
||||
new RegExp(`(?:^|\\n)\\s*(?:\\d+\\.\\s*|[-•]\\s*)\\*\\*([^*]{2,60})\\*\\*\\s*[-–—:]\\s*([^\\n]{3,120}?\\b(?:${placeTypeWords})\\b[^\\n]{0,40})`, 'gim'),
|
||||
]
|
||||
|
||||
for (const re of patterns) {
|
||||
@@ -996,12 +1044,71 @@ function extractPlacesFromPatterns(text: string): Place[] {
|
||||
}))
|
||||
}
|
||||
|
||||
function extractPlacesFromBoldPatterns(text: string): Place[] {
|
||||
const places: { name: string; desc: string; pos: number }[] = []
|
||||
const seen = new Set<string>()
|
||||
// Match any bold title followed by dash/colon and description (for place-query context)
|
||||
const re = /(?:^|\n)\s*(?:\d+\.\s*|[-•]\s*)\*\*([^*]{2,60})\*\*\s*[-–—:]\s*([^\n]{5,200})/gm
|
||||
let m: RegExpExecArray | null
|
||||
while ((m = re.exec(text)) !== null) {
|
||||
const name = m[1].trim()
|
||||
if (name.length < 2) continue
|
||||
if (/\[\[(film|song|podcast|book|tv|place)(_ext)?:/.test(name)) continue
|
||||
const key = name.toLowerCase()
|
||||
if (seen.has(key)) continue
|
||||
seen.add(key)
|
||||
places.push({ name, desc: m[2].trim(), pos: m.index })
|
||||
}
|
||||
return places
|
||||
.sort((a, b) => a.pos - b.pos)
|
||||
.map(({ name, desc }) => ({
|
||||
id: `ext-place-${name.toLowerCase().replace(/\W/g, '-')}`,
|
||||
name,
|
||||
description: desc,
|
||||
sources: [],
|
||||
}))
|
||||
}
|
||||
|
||||
export function extractAllPlaces(text: string, userQuery: string): Place[] {
|
||||
const external = extractExternalPlaces(text)
|
||||
if (external.length > 0) return external
|
||||
if (!isPlaceQuery(userQuery) && !isPlaceLikeResponse(text)) return []
|
||||
if (isNewsLikeResponse(text)) return []
|
||||
return extractPlacesFromPatterns(text)
|
||||
const fromPatterns = extractPlacesFromPatterns(text)
|
||||
// For place queries, supplement with bold-title patterns to catch items without type words
|
||||
if (isPlaceQuery(userQuery)) {
|
||||
const fromBold = extractPlacesFromBoldPatterns(text)
|
||||
const seenNames = new Set(fromPatterns.map(p => p.name.toLowerCase()))
|
||||
const extras = fromBold.filter(p => !seenNames.has(p.name.toLowerCase()))
|
||||
const combined = [...fromPatterns, ...extras]
|
||||
return combined.length > 0 ? combined : []
|
||||
}
|
||||
return fromPatterns
|
||||
}
|
||||
|
||||
// ─── Code block extraction ─────────────────────────────────────────
|
||||
|
||||
export interface CodeBlock {
|
||||
language: string
|
||||
code: string
|
||||
label?: string
|
||||
}
|
||||
|
||||
export function extractCodeBlocks(text: string): CodeBlock[] {
|
||||
const blocks: CodeBlock[] = []
|
||||
const re = /```(\w*)\n([\s\S]*?)```/g
|
||||
let m: RegExpExecArray | null
|
||||
while ((m = re.exec(text)) !== null) {
|
||||
const language = m[1] || 'text'
|
||||
const code = m[2].trimEnd()
|
||||
if (code.length < 3) continue
|
||||
// Try to find a label from the line above the code block
|
||||
const before = text.slice(Math.max(0, m.index - 200), m.index)
|
||||
const labelMatch = /(?:^|\n)\s*(?:\*\*([^*]{2,80})\*\*|#+\s+(.{2,80}))\s*\n?\s*$/.exec(before)
|
||||
const label = labelMatch?.[1] || labelMatch?.[2] || undefined
|
||||
blocks.push({ language, code, label })
|
||||
}
|
||||
return blocks
|
||||
}
|
||||
|
||||
// ─── Tag stripping ────────────────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user