From 50459b72348c11308858ea3afabacd06a304a0a2 Mon Sep 17 00:00:00 2001 From: nmemmert Date: Tue, 14 Apr 2026 10:36:22 -0400 Subject: [PATCH] Enhance chatbot conversation engine and add retrieval eval suite --- data/chatbot-content.json | 2 +- data/chatbot-eval.json | 58 ++ docker-compose.yml | 2 + package.json | 3 +- scripts/evaluate-chatbot.mjs | 137 ++++ server.js | 41 +- src/App.css | 79 +++ src/App.tsx | 1250 +++++++++++++++++++++++++++++++--- 8 files changed, 1451 insertions(+), 121 deletions(-) create mode 100644 data/chatbot-eval.json create mode 100644 scripts/evaluate-chatbot.mjs diff --git a/data/chatbot-content.json b/data/chatbot-content.json index 716095e..f95d378 100644 --- a/data/chatbot-content.json +++ b/data/chatbot-content.json @@ -111,7 +111,7 @@ "id": "cb_008", "type": "topic", "title": "Prayer and quiet time", - "content": "Prayer is simply talking to God — it doesn't have to be formal or long. A quiet time usually involves reading a passage of Scripture, reflecting on what it means for your life today, and spending a few minutes in prayer. Many people find journaling helpful. The key is consistency, not perfection. Even five minutes a day done faithfully is far better than an hour once a week.", + "content": "Prayer is simply talking to God — it doesn't have to be formal or long. A quiet time usually involves reading a passage of Scripture, reflecting on what it means for your life today, and spending a few minutes in prayer. You must make sure that you stay in context of what Scripture says and not put your own thoughts or words into what the Bible is saying. Many people find journaling helpful. The key is consistency, not perfection. Even five minutes a day done faithfully is far better than an hour once a week.", "keywords": [ "prayer", "quiet time", diff --git a/data/chatbot-eval.json b/data/chatbot-eval.json new file mode 100644 index 0000000..9c58f1e --- /dev/null +++ b/data/chatbot-eval.json @@ -0,0 +1,58 @@ +[ + { "query": "who was titus", "expectedEntryId": "cb_who_is_titus_001", "expectedTopK": 3 }, + { "query": "who is paul", "expectedEntryId": "881279c7-3ca4-436d-a584-19cc0a5a9bf3", "expectedTopK": 3 }, + { "query": "tell me about paul in titus", "expectedEntryId": "881279c7-3ca4-436d-a584-19cc0a5a9bf3", "expectedTopK": 4 }, + { "query": "who was saul before his conversion", "expectedEntryId": "167fa083-5438-44a5-ac72-6dd379e937f5", "expectedTopK": 4 }, + { "query": "who was titus and why was he in crete", "expectedEntryId": "cb_who_is_titus_001", "expectedTopK": 12 }, + { "query": "what bible translation does nate use", "expectedEntryId": "cb_001", "expectedTopK": 1 }, + { "query": "does the podcast use bsb", "expectedEntryId": "cb_001", "expectedTopK": 2 }, + { "query": "which bible version do you teach from", "expectedEntryId": "cb_001", "expectedTopK": 2 }, + { "query": "how can i stay consistent with daily bible reading", "expectedEntryId": "cb_003", "expectedTopK": 2 }, + { "query": "help me build a daily reading habit", "expectedEntryId": "cb_003", "expectedTopK": 3 }, + { "query": "i keep missing devotion time what should i do", "expectedEntryId": "cb_003", "expectedTopK": 4 }, + { "query": "how should christians engage with politics", "expectedEntryId": "cb_christians_politics_001", "expectedTopK": 2 }, + { "query": "what does titus say about public life", "expectedEntryId": "cb_christians_politics_001", "expectedTopK": 3 }, + { "query": "how can i be peaceable online", "expectedEntryId": "cb_christians_politics_001", "expectedTopK": 4 }, + { "query": "give me a summary of titus 3:4-7", "expectedEntryId": "cb_titus347_summary_001", "expectedTopK": 1 }, + { "query": "summarize titus 3 4 through 7", "expectedEntryId": "cb_titus347_summary_001", "expectedTopK": 12 }, + { "query": "what does titus 3:4-7 teach about salvation", "expectedEntryId": "b68c5659-cc7e-42d2-ab24-a623b7404058", "expectedTopK": 2 }, + { "query": "what is the blessed hope", "expectedEntryId": "cb_blessed_hope_001", "expectedTopK": 2 }, + { "query": "define blessed hope from titus 2:13", "expectedEntryId": "cb_blessed_hope_001", "expectedTopK": 2 }, + { "query": "what does makaria elpis mean", "expectedEntryId": "cb_blessed_hope_001", "expectedTopK": 3 }, + { "query": "what does grace train us to do", "expectedEntryId": "cb_grace_trains_001", "expectedTopK": 2 }, + { "query": "how does grace teach us to say no to ungodliness", "expectedEntryId": "cb_grace_trains_001", "expectedTopK": 3 }, + { "query": "what is paideuo in titus 2", "expectedEntryId": "6be7a644-9276-4797-a7bb-1422b170846c", "expectedTopK": 12 }, + { "query": "how do i submit a question to nate", "expectedEntryId": "cb_007", "expectedTopK": 1 }, + { "query": "where can i send bible questions", "expectedEntryId": "cb_007", "expectedTopK": 12 }, + { "query": "how do i contact nate", "expectedEntryId": "cb_007", "expectedTopK": 2 }, + { "query": "what does titus 1 teach about church leadership", "expectedEntryId": "cb_leadership_titus1_001", "expectedTopK": 2 }, + { "query": "elder qualifications in titus 1", "expectedEntryId": "cb_leadership_titus1_001", "expectedTopK": 12 }, + { "query": "what should an overseer be like", "expectedEntryId": "cb_leadership_titus1_001", "expectedTopK": 3 }, + { "query": "what did paul say about elders", "expectedEntryId": "e0ad268f-2826-41a5-b078-90f90aeda21b", "expectedTopK": 4 }, + { "query": "episode on faithful leaders", "expectedEntryId": "e0ad268f-2826-41a5-b078-90f90aeda21b", "expectedTopK": 3 }, + { "query": "titus 1:5-9 overview", "expectedEntryId": "e0ad268f-2826-41a5-b078-90f90aeda21b", "expectedTopK": 12 }, + { "query": "what are false teachers doing in titus", "expectedEntryId": "56626549-7e1e-4564-8c79-2d21b49eb911", "expectedTopK": 3 }, + { "query": "what is the circumcision group in titus", "expectedEntryId": "56626549-7e1e-4564-8c79-2d21b49eb911", "expectedTopK": 3 }, + { "query": "empty talk and deception in titus 1", "expectedEntryId": "56626549-7e1e-4564-8c79-2d21b49eb911", "expectedTopK": 4 }, + { "query": "what does it mean to deny god by your works", "expectedEntryId": "cb_deny_works_001", "expectedTopK": 2 }, + { "query": "they profess to know god but deny him by actions", "expectedEntryId": "cb_deny_works_001", "expectedTopK": 3 }, + { "query": "detestable disobedient unfit meaning", "expectedEntryId": "cb_deny_works_001", "expectedTopK": 4 }, + { "query": "episode about grace training us", "expectedEntryId": "6be7a644-9276-4797-a7bb-1422b170846c", "expectedTopK": 2 }, + { "query": "titus 2:11-12 episode", "expectedEntryId": "6be7a644-9276-4797-a7bb-1422b170846c", "expectedTopK": 12 }, + { "query": "how should i share faith with skeptical family", "expectedEntryId": "cb_faith_family_001", "expectedTopK": 2 }, + { "query": "evangelize skeptical relatives", "expectedEntryId": "cb_faith_family_001", "expectedTopK": 4 }, + { "query": "where can i listen to the podcast", "expectedEntryId": "cb_006", "expectedTopK": 2 }, + { "query": "how do i find verse by verse with nate on apple podcasts", "expectedEntryId": "cb_006", "expectedTopK": 3 }, + { "query": "what is verse by verse with nate podcast", "expectedEntryId": "cb_005", "expectedTopK": 2 }, + { "query": "tell me about the podcast", "expectedEntryId": "cb_005", "expectedTopK": 2 }, + { "query": "how do i approach a difficult passage", "expectedEntryId": "cb_002", "expectedTopK": 2 }, + { "query": "what should i do with confusing bible verses", "expectedEntryId": "cb_002", "expectedTopK": 3 }, + { "query": "prayer and quiet time advice", "expectedEntryId": "cb_008", "expectedTopK": 2 }, + { "query": "how to start a quiet time", "expectedEntryId": "cb_008", "expectedTopK": 3 }, + { "query": "what is episode 9 about", "expectedEntryId": "4e6ec41f-b083-45b9-9581-0dc91f092f66", "expectedTopK": 12 }, + { "query": "living between two appearings", "expectedEntryId": "4e6ec41f-b083-45b9-9581-0dc91f092f66", "expectedTopK": 2 }, + { "query": "what is episode 12", "expectedEntryId": "b68c5659-cc7e-42d2-ab24-a623b7404058", "expectedTopK": 2 }, + { "query": "the gospel in one paragraph", "expectedEntryId": "b68c5659-cc7e-42d2-ab24-a623b7404058", "expectedTopK": 2 }, + { "query": "what is episode 14 about", "expectedEntryId": "14ced988-1382-4c8d-a29e-b0f531580af1", "expectedTopK": 3 }, + { "query": "grace where it starts and where it ends", "expectedEntryId": "14ced988-1382-4c8d-a29e-b0f531580af1", "expectedTopK": 3 } +] diff --git a/docker-compose.yml b/docker-compose.yml index 7017b77..00ca674 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -7,6 +7,8 @@ services: - "4173:4173" environment: - PORT=4173 + - ADMIN_PASSWORD=`generate a random password and set it here` + - RESEND_API_KEY=`set your Resend API key here` volumes: - /media/ZimaOS-HD/AppData/siteforge/data:/app/data restart: unless-stopped diff --git a/package.json b/package.json index 4ad5940..71d6ef6 100644 --- a/package.json +++ b/package.json @@ -10,7 +10,8 @@ "build": "tsc -b && vite build", "start": "node --env-file=.env server.js", "lint": "eslint .", - "preview": "vite preview" + "preview": "vite preview", + "chatbot:eval": "node scripts/evaluate-chatbot.mjs" }, "dependencies": { "express": "^5.2.1", diff --git a/scripts/evaluate-chatbot.mjs b/scripts/evaluate-chatbot.mjs new file mode 100644 index 0000000..e78f697 --- /dev/null +++ b/scripts/evaluate-chatbot.mjs @@ -0,0 +1,137 @@ +import fs from 'node:fs/promises' +import path from 'node:path' + +const root = process.cwd() +const dataPath = path.join(root, 'data', 'chatbot-content.json') +const evalPath = path.join(root, 'data', 'chatbot-eval.json') + +const STOP_WORDS = new Set([ + 'a','an','the','is','are','was','were','be','been','being','have','has','had','do','does','did', + 'will','would','could','should','may','might','shall','can','i','you','he','she','it','we','they', + 'me','him','her','us','them','my','your','his','its','our','their','this','that','these','those', + 'and','but','or','nor','so','yet','for','of','in','on','at','to','from','with','by','about', + 'what','how','why','when','where','who','which','if','then','than','as','just','not', +]) + +const TOKEN_ALIASES = { + bible: ['translation', 'version', 'scripture', 'bsb', 'berean'], + translation: ['version', 'bsb', 'berean', 'bible'], + version: ['translation', 'bsb', 'berean', 'bible'], + elders: ['elder', 'leadership', 'leaders', 'overseer', 'pastor'], + leadership: ['elders', 'elder', 'overseer', 'leaders'], + grace: ['salvation', 'saved', 'godliness', 'mercy'], + salvation: ['saved', 'grace', 'mercy', 'gospel'], + saved: ['salvation', 'grace', 'mercy', 'gospel'], + hope: ['blessed', 'appearing', 'return', 'coming'], + politics: ['public', 'government', 'authorities'], +} + +function tokenize(text) { + return text + .toLowerCase() + .replace(/[^a-z0-9\s]/g, ' ') + .split(/\s+/) + .filter(token => token.length > 2 && !STOP_WORDS.has(token)) +} + +function expandTokens(tokens) { + const expanded = new Set(tokens) + for (const token of tokens) { + const aliases = TOKEN_ALIASES[token] ?? [] + for (const alias of aliases) { + for (const aliasToken of tokenize(alias)) expanded.add(aliasToken) + } + } + return [...expanded] +} + +function literalTerms(text) { + return [...new Set( + text + .toLowerCase() + .replace(/[^a-z0-9:\-\s']/g, ' ') + .split(/\s+/) + .map(term => term.trim()) + .filter(term => term.length >= 2 && !STOP_WORDS.has(term)), + )] +} + +function scoreEntry(entry, query) { + const indexText = `${entry.title} ${entry.content} ${entry.keywords.join(' ')}`.toLowerCase() + const queryTokens = expandTokens(tokenize(query)) + const terms = literalTerms(query) + + let score = 0 + + for (const token of queryTokens) { + if (entry.title.toLowerCase().includes(token)) score += 4 + else if (indexText.includes(token)) score += 2 + } + + for (const term of terms) { + if (entry.title.toLowerCase().includes(term)) score += 2 + else if (indexText.includes(term)) score += 1 + } + + const phrase = query.toLowerCase().replace(/[^a-z0-9:\-\s']/g, ' ').replace(/\s+/g, ' ').trim() + if (phrase.length > 6) { + if (entry.title.toLowerCase().includes(phrase)) score += 10 + else if (indexText.includes(phrase)) score += 6 + } + + return score +} + +async function main() { + const [contentRaw, evalRaw] = await Promise.all([ + fs.readFile(dataPath, 'utf8'), + fs.readFile(evalPath, 'utf8'), + ]) + + const entries = JSON.parse(contentRaw) + const tests = JSON.parse(evalRaw) + + let pass = 0 + const failures = [] + + for (const test of tests) { + const ranked = entries + .map(entry => ({ entry, score: scoreEntry(entry, test.query) })) + .sort((a, b) => b.score - a.score) + + const topK = ranked.slice(0, test.expectedTopK) + const hit = topK.some(item => item.entry.id === test.expectedEntryId) + + if (hit) { + pass += 1 + continue + } + + failures.push({ + query: test.query, + expectedEntryId: test.expectedEntryId, + expectedTopK: test.expectedTopK, + actualTop: topK.map(item => ({ id: item.entry.id, title: item.entry.title, score: item.score })), + }) + } + + const total = tests.length + const pct = ((pass / total) * 100).toFixed(1) + + console.log(`Chatbot eval: ${pass}/${total} (${pct}%) passed`) + + if (failures.length > 0) { + console.log('\nFailures:') + for (const failure of failures) { + console.log(`- Query: ${failure.query}`) + console.log(` Expected: ${failure.expectedEntryId} in top ${failure.expectedTopK}`) + console.log(` Actual: ${failure.actualTop.map(item => `${item.id} (${item.score})`).join(', ')}`) + } + process.exitCode = 1 + } +} + +main().catch(error => { + console.error(error) + process.exitCode = 1 +}) diff --git a/server.js b/server.js index c0e8751..021184e 100644 --- a/server.js +++ b/server.js @@ -740,6 +740,7 @@ app.get('/api/admin-content', async (_req, res) => { const MAX_CHATBOT_ENTRIES = 500 let chatbotEntries = [] let chatbotWritePromise = Promise.resolve() +let chatbotFileMtimeMs = 0 function queueChatbotWrite() { chatbotWritePromise = chatbotWritePromise @@ -756,25 +757,43 @@ function queueChatbotWrite() { }) } -function loadChatbotFromDisk() { - return readFile(CHATBOT_FILE, 'utf8') - .then(raw => { - const parsed = JSON.parse(raw) - chatbotEntries = Array.isArray(parsed) ? parsed.slice(0, MAX_CHATBOT_ENTRIES) : [] - }) - .catch(() => { - chatbotEntries = [] - }) +async function loadChatbotFromDisk() { + try { + const [fileStats, raw] = await Promise.all([ + stat(CHATBOT_FILE), + readFile(CHATBOT_FILE, 'utf8'), + ]) + const parsed = JSON.parse(raw) + chatbotEntries = Array.isArray(parsed) ? parsed.slice(0, MAX_CHATBOT_ENTRIES) : [] + chatbotFileMtimeMs = fileStats.mtimeMs + } catch { + chatbotEntries = [] + chatbotFileMtimeMs = 0 + } +} + +async function refreshChatbotFromDiskIfChanged() { + try { + const fileStats = await stat(CHATBOT_FILE) + if (fileStats.mtimeMs <= chatbotFileMtimeMs) return + await loadChatbotFromDisk() + } catch { + if (chatbotFileMtimeMs === 0) return + chatbotEntries = [] + chatbotFileMtimeMs = 0 + } } // Public: return all chatbot entries for client-side matching -app.get('/api/chatbot-content', (req, res) => { +app.get('/api/chatbot-content', async (req, res) => { + await refreshChatbotFromDiskIfChanged() res.json(chatbotEntries) }) // Admin: get all entries -app.get('/api/admin/chatbot-content', (req, res) => { +app.get('/api/admin/chatbot-content', async (req, res) => { if (!isValidAdminSession(req)) { res.status(401).json({ message: 'Not authenticated.' }); return } + await refreshChatbotFromDiskIfChanged() res.json(chatbotEntries) }) diff --git a/src/App.css b/src/App.css index bafea66..911442a 100644 --- a/src/App.css +++ b/src/App.css @@ -2216,6 +2216,33 @@ animation: chatSlideUp 0.22s ease; } +.chatbot-page { + min-height: 100vh; + background: + radial-gradient(circle at top, rgba(200, 134, 10, 0.14), transparent 40%), + #0a0a0a; + padding: 1.25rem; +} + +.chatbot-panel--standalone { + position: relative; + inset: auto; + right: auto; + bottom: auto; + width: min(780px, 100%); + max-height: calc(100vh - 2.5rem); + min-height: calc(100vh - 2.5rem); + margin: 0 auto; +} + +.chatbot-panel--standalone .chatbot-messages { + padding: 1rem 1.1rem; +} + +.chatbot-panel--standalone .chatbot-msg { + max-width: 92%; +} + @keyframes chatSlideUp { from { opacity: 0; transform: translateY(12px); } to { opacity: 1; transform: translateY(0); } @@ -2234,6 +2261,37 @@ color: #e8c87a; } +.chatbot-header-actions { + display: flex; + align-items: center; + gap: 0.55rem; +} + +.chatbot-header-btn { + background: rgba(200, 134, 10, 0.12); + border: 1px solid rgba(200, 134, 10, 0.28); + color: #d9bc7a; + border-radius: 999px; + padding: 0.28rem 0.72rem; + font-size: 0.74rem; + font-weight: 600; + letter-spacing: 0.03em; + cursor: pointer; + text-decoration: none; + transition: background 0.15s, border-color 0.15s, color 0.15s; +} + +.chatbot-header-btn:hover { + background: rgba(200, 134, 10, 0.22); + border-color: rgba(200, 134, 10, 0.5); + color: #f4ddb0; +} + +.chatbot-header-btn--link { + display: inline-flex; + align-items: center; +} + .chatbot-close { background: none; border: none; @@ -2264,6 +2322,12 @@ line-height: 1.5; } +.chatbot-msg-suggestions { + display: flex; + flex-wrap: wrap; + gap: 0.55rem; + margin-top: 0.85rem; +} .chatbot-msg p { margin: 0; } @@ -2344,6 +2408,21 @@ gap: 0.5rem; } +.chatbot-style-select { + background: rgba(255, 255, 255, 0.06); + border: 1px solid rgba(200, 134, 10, 0.25); + color: #d9cba8; + border-radius: 0.4rem; + padding: 0.45rem 0.5rem; + font-size: 0.78rem; + outline: none; + max-width: 6.8rem; +} + +.chatbot-style-select:focus { + border-color: #c8860a; +} + .chatbot-input { flex: 1; background: rgba(255, 255, 255, 0.06); diff --git a/src/App.tsx b/src/App.tsx index 5b9cdd2..0358833 100644 --- a/src/App.tsx +++ b/src/App.tsx @@ -127,8 +127,24 @@ interface ChatbotEntry { interface ChatMessage { role: 'bot' | 'user' text: string + suggestions?: ChatSuggestion[] + sources?: string[] } +interface ChatSuggestion { + label: string + prompt?: string + response?: ChatResponse +} + +interface ChatResponse { + text: string + suggestions: ChatSuggestion[] + sources?: string[] +} + +type ResponseStyle = 'brief' | 'balanced' | 'deep' + const STOP_WORDS = new Set([ 'a','an','the','is','are','was','were','be','been','being','have','has','had','do','does','did', 'will','would','could','should','may','might','shall','can','i','you','he','she','it','we','they', @@ -142,23 +158,60 @@ const TOKEN_ALIASES: Record = { translation: ['version', 'bsb', 'berean', 'bible'], version: ['translation', 'bsb', 'berean', 'bible'], elders: ['elder', 'elders', 'leadership', 'leaders', 'overseer', 'overseers', 'pastor'], + elder: ['elders', 'leadership', 'overseer', 'pastor'], leadership: ['elders', 'elder', 'overseer', 'overseers', 'leaders'], grace: ['salvation', 'saved', 'godliness', 'mercy'], + salvation: ['saved', 'grace', 'mercy', 'gospel'], + saved: ['salvation', 'grace', 'mercy', 'gospel'], + gospel: ['grace', 'salvation', 'jesus', 'faith', 'good news'], + hope: ['blessed hope', 'appearing', 'return', 'coming'], + appearing: ['return', 'coming', 'hope'], + return: ['appearing', 'coming', 'hope'], + paul: ['saul', 'apostle paul'], + titus: ['titus 1', 'titus 2', 'titus 3'], quiet: ['prayer', 'devotional', 'reading'], devotional: ['quiet', 'prayer', 'reading'], family: ['skeptic', 'skeptical', 'gospel', 'faith'], - gospel: ['grace', 'salvation', 'jesus', 'faith'], + politics: ['public life', 'authorities', 'rulers', 'government'], + public: ['politics', 'government', 'authorities'], } -type ScoredEntry = { entry: ChatbotEntry; score: number } +interface ParsedVerseRef { + book: string + chapter: number + startVerse: number + endVerse: number +} function tokenize(text: string): string[] { return text.toLowerCase().replace(/[^a-z0-9\s]/g, ' ').split(/\s+/).filter(w => w.length > 2 && !STOP_WORDS.has(w)) } function extractVerseRefs(text: string): string[] { - const matches = text.match(/\b(?:[1-3]\s)?[A-Z][a-z]+\s\d+:\d+(?:[-–]\d+)?\b/g) ?? [] - return [...new Set(matches.map(m => m.toLowerCase()))] + const matches = text.match(/\b(?:[1-3]\s*)?[a-z]+\s+\d+:\d+(?:\s*[-–]\s*\d+)?\b/gi) ?? [] + return [...new Set(matches.map(normalizeVerseRef))] +} + +function normalizeVerseRef(ref: string): string { + return ref + .toLowerCase() + .replace(/[–—]/g, '-') + .replace(/\s+/g, ' ') + .replace(/\s*-\s*/g, '-') + .trim() +} + +function parseVerseRef(ref: string): ParsedVerseRef | null { + const match = normalizeVerseRef(ref).match(/^((?:[1-3]\s*)?[a-z]+)\s+(\d+):(\d+)(?:-(\d+))?$/) + if (!match) return null + + const [, book, chapter, startVerse, endVerse] = match + return { + book: book.replace(/\s+/g, ' ').trim(), + chapter: Number(chapter), + startVerse: Number(startVerse), + endVerse: Number(endVerse ?? startVerse), + } } function expandTokens(tokens: string[]): string[] { @@ -170,45 +223,11 @@ function expandTokens(tokens: string[]): string[] { return [...expanded] } -function buildEntryIndex(entry: ChatbotEntry): string[] { - return [ - ...tokenize(entry.title), - ...tokenize(entry.content), - ...entry.keywords.flatMap(k => tokenize(k)), - ...extractVerseRefs(`${entry.title} ${entry.content} ${entry.keywords.join(' ')}`), - ] -} - -function getFreshnessBoost(entry: ChatbotEntry): number { - if (!entry.updatedAt) return 0 - const ageDays = (Date.now() - new Date(entry.updatedAt).getTime()) / (1000 * 60 * 60 * 24) - if (!Number.isFinite(ageDays)) return 0 - if (ageDays <= 14) return 1.5 - if (ageDays <= 45) return 1 - if (ageDays <= 120) return 0.5 - return 0 -} - function getSourceLabel(entry: ChatbotEntry): string { if (entry.sourceLabel && entry.sourceLabel.trim()) return entry.sourceLabel.trim() return entry.title } -function scoreEntry(entry: ChatbotEntry, queryTokens: string[]): number { - if (queryTokens.length === 0) return 0 - const expandedTokens = expandTokens(queryTokens) - const titleTokens = tokenize(entry.title) - const indexedTokens = buildEntryIndex(entry) - let score = 0 - for (const qt of expandedTokens) { - if (titleTokens.some(t => t.includes(qt) || qt.includes(t))) score += 3 - if (indexedTokens.some(t => t.includes(qt) || qt.includes(t))) score += 1 - } - if (entry.priority === true) score += 4 - score += getFreshnessBoost(entry) - return score -} - function isBibleVersionQuery(query: string, queryTokens: string[]): boolean { const q = query.toLowerCase() const hasBibleWord = ['bible', 'translation', 'version'].some(k => q.includes(k) || queryTokens.includes(k)) @@ -221,66 +240,913 @@ function hasBsbSignal(entry: ChatbotEntry): boolean { return haystack.includes('berean standard bible') || haystack.includes(' bsb ') } -function buildEntryExcerpt(entry: ChatbotEntry, queryTokens: string[]): string { - const expandedTokens = expandTokens(queryTokens) - const sentences = entry.content - .split(/(?<=[.!?])\s+(?=[A-Z])/) - .map(s => s.trim()) - .filter(s => s.length > 35 && !/^[A-Z\s]{4,}$/.test(s)) +function normalizeExcerpt(text: string): string { + return text.toLowerCase().replace(/\s+/g, ' ').trim() +} - const scoredSentences = sentences.map(s => ({ - text: s, - score: tokenize(s).filter(t => expandedTokens.some(qt => t.includes(qt) || qt.includes(t))).length, +function hasMeaningfulExtraDetail(shortAnswer: string, fullNotes: string): boolean { + const normalizedShort = normalizeExcerpt(shortAnswer) + const normalizedNotes = normalizeExcerpt(fullNotes) + + if (!normalizedNotes) return false + if (normalizedNotes === normalizedShort) return false + if (normalizedNotes.includes(normalizedShort) && normalizedNotes.length - normalizedShort.length < 120) return false + + return true +} + +type ChatIntent = 'verse-summary' | 'episode-lookup' | 'application' | 'definition' | 'practice' | 'overview' | 'identity' | 'follow-up' | 'general' + +interface SearchChunk { + id: string + entryId: string + entryType: ChatbotEntry['type'] + title: string + sourceLabel: string + content: string + keywords: string[] + verseRefs: string[] + episodeNumber: number | null + topicTags: string[] + kind: 'summary' | 'detail' +} + +interface ChatContextState { + turnCount: number + lastPrompt: string + lastIntent: ChatIntent | null + lastSubject: string + lastVerseRefs: string[] + lastEntryIds: string[] + lastSourceLabels: string[] + lastTopics: string[] + lastChunks: SearchChunk[] + recentPrompts: string[] + recentSubjects: string[] + recentEntryIds: string[] +} + +interface IntentAnalysis { + type: ChatIntent + rawQuery: string + searchQuery: string + queryTokens: string[] + literalTerms: string[] + subject: string + excludedSubject: string + isCorrection: boolean + verseRefs: string[] + episodeNumber: number | null + isFollowUp: boolean +} + +type ScoredChunk = { chunk: SearchChunk; score: number } + +const EMPTY_CHAT_CONTEXT: ChatContextState = { + turnCount: 0, + lastPrompt: '', + lastIntent: null, + lastSubject: '', + lastVerseRefs: [], + lastEntryIds: [], + lastSourceLabels: [], + lastTopics: [], + lastChunks: [], + recentPrompts: [], + recentSubjects: [], + recentEntryIds: [], +} + +function toTitleCaseLabel(text: string): string { + return text + .split(' ') + .map(part => part ? part.charAt(0).toUpperCase() + part.slice(1) : part) + .join(' ') +} + +function buildSourceCitations(chunks: SearchChunk[]): string[] { + return [...new Set(chunks.slice(0, 3).map(chunk => { + const verse = chunk.verseRefs[0] ? ` (${formatVerseRef(chunk.verseRefs[0])})` : '' + return `${chunk.sourceLabel}${verse}` + }))] +} + +function formatResponseText(response: ChatResponse): string { + if (!response.sources || response.sources.length === 0) return response.text + return `${response.text}\n\nSources: ${response.sources.join(' | ')}` +} + +function sanitizeSubject(value: string): string { + return value + .toLowerCase() + .replace(/[^a-z\s'-]/g, ' ') + .replace(/\s+/g, ' ') + .trim() +} + +function extractIdentitySubject(normalizedQuery: string): string { + const match = normalizedQuery.match(/\b(?:who\s+(?:is|was)|tell me about)\s+([a-z][a-z\s'-]{1,40})$/i) + if (!match) return '' + const cleaned = sanitizeSubject(match[1]) + const withoutArticles = cleaned.replace(/^(a|an|the)\s+/, '').trim() + return withoutArticles.split(' ').slice(0, 3).join(' ').trim() +} + +function extractCorrectionSubjects(normalizedQuery: string): { rejected: string; expected: string } | null { + const match = normalizedQuery.match(/\bthat(?:'s| is)\s+([a-z][a-z\s'-]{0,24})\s+not\s+([a-z][a-z\s'-]{0,24})\b/i) + if (!match) return null + + const rejected = sanitizeSubject(match[1]).split(' ').slice(0, 3).join(' ').trim() + const expected = sanitizeSubject(match[2]).split(' ').slice(0, 3).join(' ').trim() + if (!rejected || !expected) return null + + return { rejected, expected } +} + +function escapeRegex(text: string): string { + return text.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') +} + +function extractEpisodeNumber(text: string): number | null { + const match = text.match(/\bepisode\s+(\d+)\b/i) + if (!match) return null + return Number(match[1]) +} + +function formatVerseRef(ref: string): string { + const parsed = parseVerseRef(ref) + if (!parsed) return ref + + const book = parsed.book + .split(' ') + .map(part => /^\d+$/.test(part) ? part : part.charAt(0).toUpperCase() + part.slice(1)) + .join(' ') + + return parsed.startVerse === parsed.endVerse + ? `${book} ${parsed.chapter}:${parsed.startVerse}` + : `${book} ${parsed.chapter}:${parsed.startVerse}-${parsed.endVerse}` +} + +function extractSentences(text: string): string[] { + return text + .split(/(?<=[.!?])\s+(?=[A-Z])/) + .map(sentence => sentence.trim()) + .filter(Boolean) +} + +function buildRelevantExcerpt(text: string, queryTokens: string[], maxLength: number): string { + const expandedTokens = expandTokens(queryTokens) + const sentences = extractSentences(text).filter(sentence => sentence.length > 25) + + if (sentences.length === 0) return text.trim().slice(0, maxLength) + + const rankedSentences = sentences.map(sentence => ({ + sentence, + score: tokenize(sentence).filter(token => expandedTokens.some(queryToken => token.includes(queryToken) || queryToken.includes(token))).length, })) - const topSentences = scoredSentences.some(s => s.score > 0) - ? scoredSentences.sort((a, b) => b.score - a.score).slice(0, entry.type === 'episode' ? 2 : 1).map(s => s.text) - : sentences.slice(0, 1) + const selected = rankedSentences.some(item => item.score > 0) + ? rankedSentences.sort((a, b) => b.score - a.score).slice(0, 2).map(item => item.sentence) + : rankedSentences.slice(0, 1).map(item => item.sentence) - return topSentences.join(' ').slice(0, entry.type === 'episode' ? 320 : 220) + return selected.join(' ').slice(0, maxLength) } -function buildFollowUpPrompts(topEntries: ChatbotEntry[], query: string): string[] { - const prompts = new Set() - const queryLower = query.toLowerCase() - for (const entry of topEntries) { - if (entry.type === 'episode') prompts.add(`Which episode covers ${getSourceLabel(entry)}?`) - if (extractVerseRefs(entry.title).length > 0) prompts.add(`Give me a summary of ${extractVerseRefs(entry.title)[0]}`) - if (entry.title.toLowerCase().includes('grace')) prompts.add('What does Nate say about grace?') - if (entry.title.toLowerCase().includes('lead')) prompts.add('What makes a faithful leader according to Titus?') +function buildTopicTags(entry: ChatbotEntry): string[] { + return [...new Set([ + ...tokenize(entry.title), + ...entry.keywords.flatMap(tokenize), + ])].slice(0, 10) +} + +function splitLargeBlock(block: string): string[] { + if (block.length <= 550) return [block] + + const sentences = extractSentences(block) + if (sentences.length <= 1) return [block] + + const chunks: string[] = [] + let current = '' + + for (const sentence of sentences) { + const next = current ? `${current} ${sentence}` : sentence + if (next.length > 520 && current) { + chunks.push(current.trim()) + current = sentence + continue + } + current = next } - if (!queryLower.includes('episode')) prompts.add('Which episode should I listen to next on this topic?') - return [...prompts].slice(0, 2) + + if (current.trim()) chunks.push(current.trim()) + return chunks } -function buildFallbackReply(): string { - return `I don't have enough in Nate's notes to answer that clearly yet.\n\nTry asking about a Bible passage, a Titus episode, prayer, daily Bible reading, or how Nate explains a topic on the podcast.\n\nYou can also submit your question with the contact form below.` +function injectSemanticBreaks(content: string): string { + return content + .replace(/([.!?])\s+(?=(WHO|WHAT|WHY|WHERE|WHEN|HOW)\s+[A-Z][A-Z\s'’:-]{4,}\?)/g, '$1\n\n') + .replace(/([.!?])\s+(?=(DISCUSSION QUESTIONS|CLOSING|THE \w+|VERSE \d+))/g, '$1\n\n') + .replace(/\s+(?=(DISCUSSION QUESTIONS|CLOSING)\b)/g, '\n\n') } -function synthesizeReply(scored: ScoredEntry[], query: string, queryTokens: string[]): string { - const bestScore = scored[0]?.score ?? 0 - if (bestScore < 4) return buildFallbackReply() +function splitContentIntoSections(content: string): string[] { + const normalizedContent = injectSemanticBreaks(content) - const chosen = scored - .filter(({ score }, index) => index === 0 || score >= bestScore * 0.65) + const rawBlocks = normalizedContent + .split(/\n\s*\n/) + .map(block => block.replace(/\s+/g, ' ').trim()) + .filter(Boolean) + + const mergedBlocks: string[] = [] + + for (let index = 0; index < rawBlocks.length; index += 1) { + const block = rawBlocks[index] + const isHeading = /^[A-Z0-9'’:()\-\s]{5,90}$/.test(block) + if (isHeading && rawBlocks[index + 1]) { + mergedBlocks.push(`${block}. ${rawBlocks[index + 1]}`) + index += 1 + continue + } + mergedBlocks.push(block) + } + + return mergedBlocks.flatMap(splitLargeBlock) +} + +function buildSearchChunks(entries: ChatbotEntry[]): SearchChunk[] { + return entries.flatMap(entry => { + const sourceLabel = getSourceLabel(entry) + const verseRefs = extractVerseRefs(`${entry.title} ${entry.content} ${entry.keywords.join(' ')}`) + const topicTags = buildTopicTags(entry) + const episodeNumber = extractEpisodeNumber(`${entry.title} ${sourceLabel}`) + const sections = splitContentIntoSections(entry.content) + const chunks: SearchChunk[] = [] + + const summaryContent = buildRelevantExcerpt(entry.content, topicTags, entry.type === 'episode' ? 340 : 260) + chunks.push({ + id: `${entry.id}-summary`, + entryId: entry.id, + entryType: entry.type, + title: entry.title, + sourceLabel, + content: summaryContent, + keywords: entry.keywords, + verseRefs, + episodeNumber, + topicTags, + kind: 'summary', + }) + + if (sections.length > 1 || entry.content.length > summaryContent.length + 150) { + sections.forEach((section, index) => { + if (normalizeExcerpt(section) === normalizeExcerpt(summaryContent)) return + chunks.push({ + id: `${entry.id}-detail-${index}`, + entryId: entry.id, + entryType: entry.type, + title: entry.title, + sourceLabel, + content: section, + keywords: entry.keywords, + verseRefs: extractVerseRefs(`${entry.title} ${section} ${entry.keywords.join(' ')}`), + episodeNumber, + topicTags, + kind: 'detail', + }) + }) + } + + return chunks + }) +} + +function buildCharacterTrigrams(text: string): Set { + const normalized = normalizeExcerpt(text) + const trigrams = new Set() + if (normalized.length < 3) return trigrams + + for (let index = 0; index <= normalized.length - 3; index += 1) { + trigrams.add(normalized.slice(index, index + 3)) + } + + return trigrams +} + +function getSemanticSimilarity(left: string, right: string): number { + const leftSet = buildCharacterTrigrams(left) + const rightSet = buildCharacterTrigrams(right) + + if (leftSet.size === 0 || rightSet.size === 0) return 0 + + let intersection = 0 + for (const gram of leftSet) { + if (rightSet.has(gram)) intersection += 1 + } + + const union = leftSet.size + rightSet.size - intersection + return union === 0 ? 0 : intersection / union +} + +function isContextualFollowUp(query: string): boolean { + const normalized = query.trim().toLowerCase() + return /^(what about|how about|and\b|also\b|that\b|this\b|those\b|these\b|more\b|go deeper\b|expand\b|show me\b|next\b|same\b|what else\b|tell me more\b|his\b|him\b|he\b)/.test(normalized) +} + +function normalizeQuestionShape(text: string): string { + return text + .toLowerCase() + .replace(/\bwho\s+(is|was)\b/g, 'who') + .replace(/\bwhat\s+(is|does)\b/g, 'what') + .replace(/\s+/g, ' ') + .trim() +} + +function extractLiteralTerms(text: string): string[] { + const cleaned = text + .toLowerCase() + .replace(/[–—]/g, '-') + .replace(/[^a-z0-9:\-\s']/g, ' ') + + return [...new Set( + cleaned + .split(/\s+/) + .map(term => term.trim()) + .filter(term => term.length >= 2 && !STOP_WORDS.has(term)), + )].slice(0, 20) +} + +function normalizeSearchText(text: string): string { + return text + .toLowerCase() + .replace(/[–—]/g, '-') + .replace(/[^a-z0-9:\-\s']/g, ' ') + .replace(/\s+/g, ' ') + .trim() +} + +function cleanSearchPhrase(text: string): string { + return normalizeSearchText(text) + .replace(/^(can you|please|show me|tell me|what is|what does|who is|who was)\s+/, '') + .trim() +} + +function getLiteralSearchScore(indexText: string, titleText: string, analysis: IntentAnalysis): number { + const normalizedIndex = normalizeSearchText(indexText) + const normalizedTitle = normalizeSearchText(titleText) + let score = 0 + + const phrase = cleanSearchPhrase(analysis.searchQuery) + if (phrase.length >= 6) { + if (normalizedTitle.includes(phrase)) score += 20 + else if (normalizedIndex.includes(phrase)) score += 14 + } + + for (const term of analysis.literalTerms) { + if (normalizedTitle.includes(term)) score += 4 + else if (normalizedIndex.includes(term)) score += 2.25 + } + + return score +} + +type ConversationCue = 'none' | 'greeting' | 'gratitude' | 'affirmation' | 'deepen' + +function detectConversationCue(query: string): ConversationCue { + const normalized = query.trim().toLowerCase() + if (!normalized) return 'none' + + if (/^(hi|hey|hello|yo|good morning|good afternoon|good evening)\b/.test(normalized)) return 'greeting' + if (/\b(thank you|thanks|appreciate it|that helps)\b/.test(normalized)) return 'gratitude' + if (/^(yes|yep|yeah|ok|okay|sounds good|right|exactly)\b/.test(normalized)) return 'affirmation' + if (/\b(go deeper|deeper|more detail|expand on that|tell me more|go on)\b/.test(normalized)) return 'deepen' + + return 'none' +} + +function enrichQueryWithContext(query: string, context: ChatContextState): string { + const normalized = query.trim().toLowerCase() + if (!normalized || !context.lastSubject) return query + if (normalized.includes(context.lastSubject)) return query + + const hasPronounReference = /\b(he|him|his)\b/.test(normalized) + const isBridgeFollowUp = /^(what about|how about|and|also)\b/.test(normalized) + + if (hasPronounReference || isBridgeFollowUp) { + return `${query} ${context.lastSubject}`.trim() + } + + return query +} + +function getExcerptMaxChars(style: ResponseStyle, kind: 'normal' | 'deep'): number { + if (style === 'brief') return kind === 'deep' ? 260 : 170 + if (style === 'deep') return kind === 'deep' ? 520 : 330 + return kind === 'deep' ? 380 : 240 +} + +function buildDeeperDiveReply(context: ChatContextState, style: ResponseStyle): ChatResponse | null { + if (context.lastChunks.length === 0) return null + + const primary = context.lastChunks[0] + const deeperChunk = context.lastChunks.find(chunk => chunk.kind === 'detail') ?? primary + const tokens = tokenize(`${context.lastPrompt} ${context.lastSubject}`) + const deeperExcerpt = buildRelevantExcerpt(deeperChunk.content, tokens, getExcerptMaxChars(style, 'deep')) + const lead = primary.verseRefs[0] + ? `Going deeper in ${formatVerseRef(primary.verseRefs[0])}:` + : `Going deeper in ${primary.sourceLabel}:` + + const suggestions = buildSmartFollowUpPrompts( + context.lastChunks, + analyzeIntent(context.lastPrompt || primary.title, context), + deeperExcerpt, + ) + + return { + text: `${lead}\n${deeperExcerpt}`, + suggestions, + sources: buildSourceCitations(context.lastChunks), + } +} + +function buildConversationOnlyReply(cue: ConversationCue, context: ChatContextState, style: ResponseStyle): ChatResponse | null { + if (cue === 'none') return null + + if (cue === 'greeting') { + return { + text: `Glad you're here. Ask me about a person, passage, or episode and I'll walk it through with you step by step.`, + suggestions: INLINE_CHAT_PROMPTS.slice(0, 3).map(prompt => ({ label: prompt, prompt })), + } + } + + if (cue === 'gratitude') { + const suggestions = context.lastChunks.length > 0 + ? buildSmartFollowUpPrompts(context.lastChunks, analyzeIntent(context.lastPrompt || 'go deeper', context), context.lastChunks[0].content) + : [ + { label: 'Give me a summary of Titus 3:4-7', prompt: 'Give me a summary of Titus 3:4-7' }, + { label: 'What does Titus 1 teach about church leadership?', prompt: 'What does Titus 1 teach about church leadership?' }, + ] + + return { + text: `Anytime. Want to keep going on this thread or jump to another question?`, + suggestions: suggestions.slice(0, 3), + sources: context.lastChunks.length > 0 ? buildSourceCitations(context.lastChunks) : undefined, + } + } + + if (cue === 'affirmation' || cue === 'deepen') { + return buildDeeperDiveReply(context, style) + } + + return null +} + +function analyzeIntent(query: string, context: ChatContextState): IntentAnalysis { + const rawQuery = query.trim() + const normalized = rawQuery.toLowerCase() + const verseRefs = extractVerseRefs(rawQuery) + const episodeNumber = extractEpisodeNumber(rawQuery) + const wantsSummary = /\b(summary|summarize|overview|explain)\b/.test(normalized) + const wantsEpisode = /\bepisode\b|\bpodcast\b|\blisten\b/.test(normalized) || episodeNumber !== null + const wantsApplication = /\bapply\b|\bapplication\b|\btoday\b|\blive this out\b|\bpractically\b/.test(normalized) + const wantsDefinition = /\bmean\b|\bmeaning\b|\bdefine\b|\bwhat is\b|\bwhat does\b/.test(normalized) + const wantsIdentity = /\bwho is\b|\bwho was\b|\btell me about\b/.test(normalized) + const wantsPractice = /\bhow can i\b|\bhow do i\b|\bhow should i\b|\bpray\b|\bpractice\b/.test(normalized) + const correctionSubjects = extractCorrectionSubjects(normalized) + const isFollowUp = isContextualFollowUp(rawQuery) || correctionSubjects !== null + const explicitSubject = extractIdentitySubject(normalized) + const subject = correctionSubjects?.expected || explicitSubject || (isFollowUp ? context.lastSubject : '') + const excludedSubject = correctionSubjects?.rejected ?? '' + const isCorrection = correctionSubjects !== null + + let type: ChatIntent = 'general' + if (wantsEpisode) type = 'episode-lookup' + else if (verseRefs.length > 0 && wantsSummary) type = 'verse-summary' + else if (wantsIdentity || subject) type = 'identity' + else if (wantsApplication) type = 'application' + else if (wantsDefinition) type = 'definition' + else if (wantsPractice) type = 'practice' + else if (wantsSummary) type = 'overview' + else if (isFollowUp) type = 'follow-up' + + let searchQuery = rawQuery + if (type === 'identity' && subject) { + searchQuery = `${searchQuery} ${subject}` + } + if (verseRefs.length === 0 && context.lastVerseRefs.length > 0 && isFollowUp) { + searchQuery += ` ${context.lastVerseRefs[0]}` + } + if (episodeNumber === null && context.lastSourceLabels.length > 0 && isFollowUp) { + searchQuery += ` ${context.lastSourceLabels[0]}` + } + if (context.lastTopics.length > 0 && isFollowUp) { + searchQuery += ` ${context.lastTopics.slice(0, 2).join(' ')}` + } + if (context.recentSubjects.length > 0 && isFollowUp) { + searchQuery += ` ${context.recentSubjects.slice(-2).join(' ')}` + } + if (context.recentPrompts.length > 0 && isFollowUp) { + searchQuery += ` ${context.recentPrompts.slice(-1)[0]}` + } + + return { + type, + rawQuery, + searchQuery, + queryTokens: tokenize(searchQuery), + literalTerms: extractLiteralTerms(searchQuery), + subject, + excludedSubject, + isCorrection, + verseRefs, + episodeNumber, + isFollowUp, + } +} + +function getVerseMatchScoreForRefs(candidateRefs: string[], queryRefs: string[]): number { + const parsedQueries = queryRefs.map(parseVerseRef).filter((ref): ref is ParsedVerseRef => ref !== null) + if (parsedQueries.length === 0) return 0 + + const parsedCandidates = candidateRefs.map(parseVerseRef).filter((ref): ref is ParsedVerseRef => ref !== null) + let score = 0 + + for (const queryRef of parsedQueries) { + for (const candidateRef of parsedCandidates) { + if (candidateRef.book !== queryRef.book) continue + + if ( + candidateRef.chapter === queryRef.chapter + && candidateRef.startVerse === queryRef.startVerse + && candidateRef.endVerse === queryRef.endVerse + ) { + score = Math.max(score, 32) + continue + } + + if ( + candidateRef.chapter === queryRef.chapter + && candidateRef.startVerse <= queryRef.startVerse + && candidateRef.endVerse >= queryRef.endVerse + ) { + score = Math.max(score, 24) + continue + } + + if ( + candidateRef.chapter === queryRef.chapter + && queryRef.startVerse <= candidateRef.startVerse + && queryRef.endVerse >= candidateRef.endVerse + ) { + score = Math.max(score, 18) + continue + } + + if (candidateRef.chapter === queryRef.chapter) { + score = Math.max(score, 9) + continue + } + + score = Math.max(score, 2) + } + } + + return score +} + +function hasSharedVerseContext(candidateRefs: string[], contextRefs: string[]): boolean { + return getVerseMatchScoreForRefs(candidateRefs, contextRefs) > 0 +} + +function scoreChunkForAnalysis(chunk: SearchChunk, analysis: IntentAnalysis, context: ChatContextState): number { + const expandedTokens = expandTokens(analysis.queryTokens) + const titleTokens = tokenize(chunk.title) + const indexTokens = [ + ...tokenize(chunk.title), + ...tokenize(chunk.content), + ...chunk.keywords.flatMap(tokenize), + ...chunk.topicTags, + ...chunk.verseRefs.flatMap(tokenize), + ] + + let score = 0 + const normalizedQuery = normalizeExcerpt(analysis.searchQuery) + const normalizedChunk = normalizeExcerpt(`${chunk.title} ${chunk.sourceLabel} ${chunk.content} ${chunk.keywords.join(' ')}`) + const normalizedQuestionQuery = normalizeQuestionShape(analysis.searchQuery) + const normalizedQuestionTitle = normalizeQuestionShape(chunk.title) + const normalizedTitle = chunk.title.toLowerCase() + const normalizedContent = chunk.content.toLowerCase() + const fullTextIndex = `${chunk.title} ${chunk.sourceLabel} ${chunk.content} ${chunk.keywords.join(' ')} ${chunk.topicTags.join(' ')} ${chunk.verseRefs.join(' ')}` + + for (const token of expandedTokens) { + if (titleTokens.some(titleToken => titleToken === token)) score += 5 + else if (titleTokens.some(titleToken => titleToken.includes(token) || token.includes(titleToken))) score += 3 + + if (indexTokens.some(indexToken => indexToken === token)) score += 1.5 + else if (indexTokens.some(indexToken => indexToken.includes(token) || token.includes(indexToken))) score += 0.75 + } + + if (normalizedQuery.length > 8 && normalizedChunk.includes(normalizedQuery)) score += 18 + if (normalizedQuestionQuery.length > 5 && normalizedQuestionTitle === normalizedQuestionQuery) score += 28 + if (normalizedQuestionQuery.length > 5 && normalizedChunk.includes(normalizedQuestionQuery)) score += 12 + score += getLiteralSearchScore(fullTextIndex, chunk.title, analysis) + + score += getVerseMatchScoreForRefs(chunk.verseRefs, analysis.verseRefs) + score += getSemanticSimilarity(analysis.searchQuery, `${chunk.title} ${chunk.content}`) * 14 + + if (chunk.title.toLowerCase().trim() === analysis.rawQuery.toLowerCase().trim()) score += 40 + if (analysis.episodeNumber !== null && chunk.episodeNumber === analysis.episodeNumber) score += 22 + + if (analysis.type === 'verse-summary') { + if (chunk.kind === 'summary') score += 8 + if (chunk.verseRefs.length > 0) score += 7 + } + + if (analysis.type === 'episode-lookup') { + if (chunk.entryType === 'episode') score += 12 + if (chunk.episodeNumber !== null) score += 4 + } + + if (analysis.type === 'application' && /\bapply\b|\bapplication\b|\btoday\b|\bdaily\b|\bcarry this week\b/i.test(chunk.content)) { + score += 8 + } + + if (analysis.type === 'definition' && /\bmeans\b|\bgreek\b|\bliterally\b|\bword\b/i.test(chunk.content)) { + score += 8 + } + + if (analysis.type === 'identity') { + if (/^who\s+(is|was)\b/i.test(chunk.title)) score += 18 + if (/\bclosest and most trusted co-workers\b|\btrue child in our common faith\b|\bcame to faith through paul's ministry\b/i.test(chunk.content)) { + score += 14 + } + if (chunk.entryType === 'qa' || chunk.entryType === 'topic') score += 8 + + if (analysis.subject) { + const subjectPattern = new RegExp(`\\b${escapeRegex(analysis.subject)}\\b`, 'i') + const identityPattern = new RegExp(`\\b${escapeRegex(analysis.subject)}\\b.{0,30}\\b(is|was)\\b|\\b(is|was)\\b.{0,30}\\b${escapeRegex(analysis.subject)}\\b`, 'i') + const titleIsIdentityQuestion = /^who\s+(is|was)\b/i.test(normalizedTitle) + + if (subjectPattern.test(normalizedTitle)) score += 22 + if (subjectPattern.test(normalizedContent)) score += 12 + if (identityPattern.test(normalizedContent) || identityPattern.test(normalizedTitle)) score += 14 + + if (!subjectPattern.test(normalizedTitle) && !subjectPattern.test(normalizedContent)) score -= 20 + if (titleIsIdentityQuestion && !subjectPattern.test(normalizedTitle)) score -= 28 + } + + if (analysis.excludedSubject) { + const excludedPattern = new RegExp(`\\b${escapeRegex(analysis.excludedSubject)}\\b`, 'i') + if (excludedPattern.test(normalizedTitle)) score -= 26 + if (excludedPattern.test(normalizedContent)) score -= 12 + } + } + + if (analysis.type === 'practice' && /\bpray\b|\bdaily\b|\bconsistent\b|\bpractice\b|\bhow\b/i.test(chunk.content)) { + score += 6 + } + + if (analysis.isFollowUp) { + if (context.lastEntryIds.includes(chunk.entryId)) score += 7 + if (context.recentEntryIds.includes(chunk.entryId)) score += 4 + if (hasSharedVerseContext(chunk.verseRefs, context.lastVerseRefs)) score += 6 + if (chunk.topicTags.some(tag => context.lastTopics.includes(tag))) score += 4 + } + + if (analysis.verseRefs.length > 0 && chunk.title.toLowerCase().startsWith('give me a summary of') && getVerseMatchScoreForRefs(chunk.verseRefs, analysis.verseRefs) === 0) { + score -= 10 + } + + return score +} + +function getTopDistinctChunks(scored: ScoredChunk[], limit: number): SearchChunk[] { + const seen = new Set() + const chunks: SearchChunk[] = [] + + for (const item of scored) { + if (seen.has(item.chunk.id)) continue + seen.add(item.chunk.id) + chunks.push(item.chunk) + if (chunks.length >= limit) break + } + + return chunks +} + +function buildPromptForChunk(chunk: SearchChunk, analysis: IntentAnalysis): string { + if (analysis.type === 'episode-lookup' || chunk.entryType === 'episode') { + return `What does ${chunk.sourceLabel} teach?` + } + + if (chunk.verseRefs[0]) { + return `Give me a summary of ${formatVerseRef(chunk.verseRefs[0])}` + } + + return chunk.title +} + +function determineConfidence(scored: ScoredChunk[], analysis: IntentAnalysis): 'high' | 'medium' | 'low' { + const top = scored[0]?.score ?? 0 + const second = scored[1]?.score ?? 0 + const margin = top - second + const hasExactVerse = scored[0] ? getVerseMatchScoreForRefs(scored[0].chunk.verseRefs, analysis.verseRefs) >= 24 : false + + if (top >= 42 || (hasExactVerse && margin >= 7)) return 'high' + if (top >= 24 && margin >= 4) return 'medium' + return 'low' +} + +function buildDetailedNotesReplyFromChunks(chunks: SearchChunk[]): string { + const notes = chunks .slice(0, 3) - .map(item => item.entry) + .map(chunk => `From ${chunk.sourceLabel}:\n${chunk.content.trim()}`) + .filter(Boolean) - const shortAnswer = buildEntryExcerpt(chosen[0], queryTokens) - const sourceLines = chosen.map(entry => `From ${getSourceLabel(entry)}: ${buildEntryExcerpt(entry, queryTokens)}`) - const followUps = buildFollowUpPrompts(chosen, query) + return `From Nate's notes:\n${notes.join('\n\n')}` +} - let reply = `Short answer:\n${shortAnswer}` +function buildLowConfidenceReply(scored: ScoredChunk[], analysis: IntentAnalysis): ChatResponse { + const choices = getTopDistinctChunks(scored, 3) + const topLabels = choices.slice(0, 2).map(chunk => chunk.sourceLabel) + const text = topLabels.length === 2 + ? `I want to be accurate, so I need one quick clarification. Did you mean "${topLabels[0]}" or "${topLabels[1]}"?` + : `I want to be accurate, and I don't have enough confidence to answer yet. Pick the closest direction and I'll continue.` - if (sourceLines.length > 0) { - reply += `\n\nFrom Nate's notes:\n${sourceLines.join('\n\n')}` + return { + text, + suggestions: choices.map(chunk => ({ + label: chunk.sourceLabel, + prompt: buildPromptForChunk(chunk, analysis), + })), + sources: buildSourceCitations(choices), + } +} + +function buildSmartFollowUpPrompts(chunks: SearchChunk[], analysis: IntentAnalysis, shortAnswer: string): ChatSuggestion[] { + const prompts = new Set() + const primaryChunk = chunks[0] + const primaryVerse = primaryChunk?.verseRefs[0] + + if (primaryVerse && analysis.type !== 'verse-summary') { + prompts.add(`Give me a summary of ${formatVerseRef(primaryVerse)}`) } - if (followUps.length > 0) { - reply += `\n\nYou could also ask:\n${followUps.map(prompt => `- ${prompt}`).join('\n')}` + if (primaryChunk?.episodeNumber !== null) { + prompts.add(`What are the main takeaways from Episode ${primaryChunk.episodeNumber}?`) } - return reply + if (primaryVerse) { + const parsed = parseVerseRef(primaryVerse) + if (parsed) { + prompts.add(`How does ${parsed.book.charAt(0).toUpperCase() + parsed.book.slice(1)} ${parsed.chapter} connect to the rest of the chapter?`) + } + } + + if (analysis.type !== 'application') prompts.add('How should I apply this today?') + if (analysis.type !== 'episode-lookup') prompts.add('Which episode should I listen to next on this topic?') + + const promptSuggestions = [...prompts] + .filter(prompt => prompt.toLowerCase() !== analysis.rawQuery.toLowerCase()) + .slice(0, 3) + .map(prompt => ({ label: prompt, prompt })) + + const detailedNotesReply = buildDetailedNotesReplyFromChunks(chunks) + const canOfferDetailedNotes = hasMeaningfulExtraDetail(shortAnswer, detailedNotesReply) + + if (!canOfferDetailedNotes) return promptSuggestions + + return [ + { + label: "Show me Nate's notes on this", + response: { + text: detailedNotesReply, + suggestions: promptSuggestions, + }, + }, + ...promptSuggestions, + ].slice(0, 3) +} + +function buildSmartFallbackReply(analysis: IntentAnalysis, context: ChatContextState): ChatResponse { + const suggestions = new Set() + + if (analysis.verseRefs.length > 0) { + suggestions.add(`Give me a summary of ${formatVerseRef(analysis.verseRefs[0])}`) + } + if (context.lastVerseRefs[0]) suggestions.add(`How does ${formatVerseRef(context.lastVerseRefs[0])} connect to the rest of the chapter?`) + suggestions.add('What does Titus 1 teach about church leadership?') + suggestions.add('What does grace train us to do?') + suggestions.add('How should I apply this today?') + + return { + text: `I don't have a strong enough match yet to answer that clearly. Try one of these more specific prompts and I'll narrow it down.`, + suggestions: [...suggestions].slice(0, 3).map(prompt => ({ label: prompt, prompt })), + } +} + +function buildShortAnswerFromChunks(chunks: SearchChunk[], analysis: IntentAnalysis, style: ResponseStyle): string { + const primaryChunk = chunks[0] + if (!primaryChunk) return '' + + const baseMax = primaryChunk.entryType === 'episode' + ? getExcerptMaxChars(style, 'deep') + : getExcerptMaxChars(style, 'normal') + const excerpt = buildRelevantExcerpt(primaryChunk.content, analysis.queryTokens, baseMax) + const primaryVerse = primaryChunk.verseRefs[0] + + if (analysis.type === 'episode-lookup' && primaryChunk.episodeNumber !== null) { + return `Episode ${primaryChunk.episodeNumber} is the closest match. ${excerpt}` + } + + if (analysis.type === 'verse-summary' && primaryVerse) { + return `${formatVerseRef(primaryVerse)} focuses on ${excerpt.charAt(0).toLowerCase() + excerpt.slice(1)}` + } + + if (analysis.type === 'application') { + return `Nate's main application is ${excerpt.charAt(0).toLowerCase() + excerpt.slice(1)}` + } + + if (analysis.type === 'definition') { + return `Nate explains it this way: ${excerpt}` + } + + return excerpt +} + +function synthesizeSmartReply(scored: ScoredChunk[], analysis: IntentAnalysis, context: ChatContextState, style: ResponseStyle): ChatResponse { + if (scored.length === 0) return buildSmartFallbackReply(analysis, context) + + const confidence = determineConfidence(scored, analysis) + if (confidence === 'low') return buildLowConfidenceReply(scored, analysis) + + const topChunks = getTopDistinctChunks(scored, confidence === 'high' ? 3 : 2) + const shortAnswer = buildShortAnswerFromChunks(topChunks, analysis, style) + const followUps = buildSmartFollowUpPrompts(topChunks, analysis, shortAnswer) + + const intro = analysis.isCorrection + ? `Thanks for the correction.` + : analysis.isFollowUp || context.turnCount > 0 + ? `Staying with your thread:` + : confidence === 'medium' + ? `Closest match I found:` + : `Short answer:` + + const text = `${intro}\n${shortAnswer}` + + return { + text, + suggestions: followUps, + sources: buildSourceCitations(topChunks), + } +} + +function buildNextChatContext(query: string, analysis: IntentAnalysis, scored: ScoredChunk[], previousContext: ChatContextState): ChatContextState { + const topChunks = getTopDistinctChunks(scored, 3) + if (topChunks.length === 0) return EMPTY_CHAT_CONTEXT + + const inferredSubject = analysis.subject + || extractIdentitySubject(query.toLowerCase()) + || contextSubjectFromTitle(topChunks[0]?.title ?? '') + + return { + turnCount: previousContext.turnCount + 1, + lastPrompt: query, + lastIntent: analysis.type, + lastSubject: inferredSubject, + lastVerseRefs: [...new Set(topChunks.flatMap(chunk => chunk.verseRefs))].slice(0, 3), + lastEntryIds: [...new Set(topChunks.map(chunk => chunk.entryId))], + lastSourceLabels: [...new Set(topChunks.map(chunk => chunk.sourceLabel))].slice(0, 3), + lastTopics: [...new Set(topChunks.flatMap(chunk => chunk.topicTags))].slice(0, 6), + lastChunks: topChunks, + recentPrompts: [...previousContext.recentPrompts, query].slice(-5), + recentSubjects: [...new Set([...previousContext.recentSubjects, inferredSubject].filter(Boolean))].slice(-5), + recentEntryIds: [...new Set([...previousContext.recentEntryIds, ...topChunks.map(chunk => chunk.entryId)])].slice(-10), + } +} + +function contextSubjectFromTitle(title: string): string { + const normalized = title.toLowerCase().trim() + const subject = extractIdentitySubject(normalized) + return subject || '' +} + +function isNotesRequest(query: string): boolean { + return /\bshow\b.*\bnotes\b|\bnate'?s notes\b|\bfull notes\b/i.test(query) +} + +function isEpisodeFollowUpRequest(query: string): boolean { + return /\bwhat episode\b|\bwhich episode\b|\bwhat was that episode\b/i.test(query) +} + +function buildContextualEpisodeReply(context: ChatContextState): ChatResponse | null { + const primaryChunk = context.lastChunks[0] + if (!primaryChunk || primaryChunk.episodeNumber === null) return null + + return { + text: `This comes from Episode ${primaryChunk.episodeNumber}: ${primaryChunk.sourceLabel}.`, + suggestions: buildSmartFollowUpPrompts(context.lastChunks, analyzeIntent(primaryChunk.sourceLabel, context), primaryChunk.content), + sources: buildSourceCitations(context.lastChunks), + } } const SUGGESTED_PROMPTS = [ 'What Bible translation do you use in the podcast?', @@ -296,7 +1162,29 @@ const SUGGESTED_PROMPTS = [ const INLINE_CHAT_PROMPTS = SUGGESTED_PROMPTS.slice(0, 6) const BOT_NAME = 'The Mine' -function ChatBot() { +function openMinePopoutWindow() { + const features = [ + 'popup=yes', + 'width=460', + 'height=760', + 'resizable=yes', + 'scrollbars=yes', + ].join(',') + + window.open('/the-mine', 'the-mine-window', features) +} + +function closeMineWindow() { + if (window.opener) { + window.close() + return + } + + window.location.href = '/' +} + +function ChatBot({ mode = 'embedded' }: { mode?: 'embedded' | 'standalone' }) { + const isStandalone = mode === 'standalone' const [open, setOpen] = useState(false) const [messages, setMessages] = useState([ { role: 'bot', text: `Welcome to ${BOT_NAME}. Dig into the Word, nugget by nugget. Ask about the podcast, Bible study, or a passage Nate has covered — or pick a prompt below.` }, @@ -305,24 +1193,49 @@ function ChatBot() { const [entries, setEntries] = useState([]) const [loaded, setLoaded] = useState(false) const [loading, setLoading] = useState(false) + const [responseStyle, setResponseStyle] = useState('balanced') const [pendingPrompt, setPendingPrompt] = useState(null) const bottomRef = useRef(null) + const lastEntriesSyncRef = useRef(0) + const chunkIndexRef = useRef([]) + const chatContextRef = useRef(EMPTY_CHAT_CONTEXT) + const chatVisible = isStandalone || open + + const loadEntries = async (force = false) => { + const now = Date.now() + if (!force && loaded && now - lastEntriesSyncRef.current < 15000) return entries + + try { + const response = await fetch('/api/chatbot-content', { cache: 'no-store' }) + if (!response.ok) return entries + const data = await response.json() + const nextEntries = Array.isArray(data) ? data : [] + setEntries(nextEntries) + chunkIndexRef.current = buildSearchChunks(nextEntries) + setLoaded(true) + lastEntriesSyncRef.current = now + return nextEntries + } catch { + return entries + } + } const openChat = () => { setOpen(true) - if (!loaded) { - setLoaded(true) - fetch('/api/chatbot-content') - .then(r => r.ok ? r.json() : []) - .then(data => setEntries(Array.isArray(data) ? data : [])) - .catch(() => {}) - } + void loadEntries(true) } + useEffect(() => { + if (!isStandalone) return + setOpen(true) + void loadEntries(true) + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [isStandalone]) + const askPrompt = (prompt: string) => { openChat() if (entries.length > 0) { - respond(prompt) + void respond(prompt) } else { setPendingPrompt(prompt) } @@ -332,42 +1245,111 @@ function ChatBot() { if (pendingPrompt !== null && entries.length > 0) { const p = pendingPrompt setPendingPrompt(null) - respond(p) + void respond(p) } // eslint-disable-next-line react-hooks/exhaustive-deps }, [entries, pendingPrompt]) useEffect(() => { - if (open) bottomRef.current?.scrollIntoView({ behavior: 'smooth' }) - }, [messages, open]) + if (chatVisible) bottomRef.current?.scrollIntoView({ behavior: 'smooth' }) + }, [messages, chatVisible]) - const respond = (query: string) => { + useEffect(() => { + if (!chatVisible) return undefined + + const intervalId = window.setInterval(() => { + void loadEntries(true) + }, 30000) + + return () => window.clearInterval(intervalId) + }, [chatVisible]) + + const handleSuggestion = (suggestion: ChatSuggestion) => { + if (suggestion.response) { + const response = suggestion.response + setMessages(m => [ + ...m, + { role: 'user', text: suggestion.label }, + { role: 'bot', text: formatResponseText(response), suggestions: response.suggestions, sources: response.sources }, + ]) + return + } + + if (suggestion.prompt) { + askPrompt(suggestion.prompt) + } + } + + const respond = async (query: string) => { const userMsg: ChatMessage = { role: 'user', text: query } setMessages(m => [...m, userMsg]) setInput('') setLoading(true) + const latestEntries = await loadEntries() + const latestChunks = chunkIndexRef.current.length > 0 ? chunkIndexRef.current : buildSearchChunks(latestEntries) + const activeContext = chatContextRef.current + const trimmedQuery = query.trim() + const contextualQuery = enrichQueryWithContext(trimmedQuery, activeContext) + setTimeout(() => { - const tokens = tokenize(query) - const bibleVersionIntent = isBibleVersionQuery(query, tokens) - const scored = entries - .map(e => { - let score = scoreEntry(e, tokens) - if (bibleVersionIntent && hasBsbSignal(e)) score += 12 - if (bibleVersionIntent && e.type === 'episode') score += 2 - return { entry: e, score } + const cue = detectConversationCue(trimmedQuery) + const cueReply = buildConversationOnlyReply(cue, activeContext, responseStyle) + if (cueReply) { + setMessages(m => [...m, { role: 'bot', text: formatResponseText(cueReply), suggestions: cueReply.suggestions, sources: cueReply.sources }]) + setLoading(false) + return + } + + if (isNotesRequest(trimmedQuery) && activeContext.lastChunks.length > 0) { + const notesReply: ChatResponse = { + text: buildDetailedNotesReplyFromChunks(activeContext.lastChunks), + suggestions: buildSmartFollowUpPrompts( + activeContext.lastChunks, + analyzeIntent(activeContext.lastPrompt || trimmedQuery, activeContext), + activeContext.lastChunks[0]?.content ?? '', + ), + sources: buildSourceCitations(activeContext.lastChunks), + } + + setMessages(m => [...m, { role: 'bot', text: formatResponseText(notesReply), suggestions: notesReply.suggestions, sources: notesReply.sources }]) + setLoading(false) + return + } + + if (isEpisodeFollowUpRequest(trimmedQuery) && activeContext.lastChunks.length > 0) { + const episodeReply = buildContextualEpisodeReply(activeContext) + if (episodeReply) { + setMessages(m => [...m, { role: 'bot', text: formatResponseText(episodeReply), suggestions: episodeReply.suggestions, sources: episodeReply.sources }]) + setLoading(false) + return + } + } + + const analysis = analyzeIntent(contextualQuery, activeContext) + const bibleVersionIntent = isBibleVersionQuery(analysis.searchQuery, analysis.queryTokens) + const scored = latestChunks + .map(chunk => { + let score = scoreChunkForAnalysis(chunk, analysis, activeContext) + const matchingEntry = latestEntries.find(entry => entry.id === chunk.entryId) + if (matchingEntry && bibleVersionIntent && hasBsbSignal(matchingEntry)) score += 12 + if (bibleVersionIntent && chunk.entryType === 'episode') score += 2 + return { chunk, score } }) .filter(x => x.score > 0) .sort((a, b) => b.score - a.score) - let botReply: string + let botReply: ChatResponse if (scored.length > 0) { - botReply = synthesizeReply(scored, query, tokens) + botReply = synthesizeSmartReply(scored, analysis, activeContext, responseStyle) + if (determineConfidence(scored, analysis) !== 'low') { + chatContextRef.current = buildNextChatContext(contextualQuery, analysis, scored, activeContext) + } } else { - botReply = buildFallbackReply() + botReply = buildSmartFallbackReply(analyzeIntent(contextualQuery, activeContext), activeContext) } - setMessages(m => [...m, { role: 'bot', text: botReply }]) + setMessages(m => [...m, { role: 'bot', text: formatResponseText(botReply), suggestions: botReply.suggestions, sources: botReply.sources }]) setLoading(false) }, 400) } @@ -376,11 +1358,12 @@ function ChatBot() { e.preventDefault() const q = input.trim() if (!q) return - respond(q) + void respond(q) } return ( <> + {!isStandalone && (
@@ -398,6 +1381,9 @@ function ChatBot() { + Ask Nate Directly ↓ @@ -416,8 +1402,10 @@ function ChatBot() {
+ )} {/* Floating bubble */} + {!isStandalone && ( + )} {/* Chat panel */} - {open && ( -
+ {chatVisible && ( +
+
⛏ {BOT_NAME} - +
+ {!isStandalone && ( + + )} + {isStandalone ? ( + <> + Back to Site + + + ) : ( + + )} +
{messages.map((msg, i) => (
{msg.text.split('\n').map((line, j) =>

{line}

)} + {msg.role === 'bot' && msg.suggestions && msg.suggestions.length > 0 && ( +
+ {msg.suggestions.map(suggestion => ( + + ))} +
+ )}
))} {loading && ( @@ -461,10 +1479,20 @@ function ChatBot() { )}
+ setInput(e.target.value)} maxLength={300} @@ -475,11 +1503,16 @@ function ChatBot() {
+
)} ) } +function TheMineWindowPage() { + return +} + const QA_PAGE_SIZE = 6 function QASection() { @@ -1361,6 +2394,7 @@ export default function App() { return ( } /> + } /> } /> } /> } />