195 lines
5.7 KiB
JavaScript
195 lines
5.7 KiB
JavaScript
import { execFileSync } from 'node:child_process'
|
|
import { randomUUID } from 'node:crypto'
|
|
import { promises as fs } from 'node:fs'
|
|
import path from 'node:path'
|
|
|
|
const ROOT = '/Users/nate.emmert/Documents/github/Siteforge'
|
|
const DOCS_DIR = path.join(ROOT, 'Verse by Verse with Nate Complete Series')
|
|
const CHATBOT_FILE = path.join(ROOT, 'data', 'chatbot-content.json')
|
|
|
|
const STOP_WORDS = new Set([
|
|
'the', 'and', 'for', 'that', 'with', 'this', 'from', 'your', 'you', 'are', 'but', 'not', 'have',
|
|
'has', 'was', 'were', 'his', 'her', 'our', 'their', 'into', 'about', 'what', 'when', 'where',
|
|
'which', 'will', 'just', 'they', 'them', 'then', 'than', 'how', 'why', 'can', 'all', 'through',
|
|
])
|
|
|
|
function parseEpisodeNumber(filePath) {
|
|
const match = path.basename(filePath).match(/Episode(\d+)/i)
|
|
return match ? Number(match[1]) : null
|
|
}
|
|
|
|
function getVariantRank(filePath) {
|
|
const name = path.basename(filePath).toLowerCase()
|
|
let score = 0
|
|
if (name.includes('expanded')) score += 30
|
|
if (name.includes('updated')) score += 20
|
|
if (!name.includes('expanded') && !name.includes('updated')) score += 10
|
|
if (filePath.includes(`${path.sep}Done${path.sep}Old${path.sep}`)) score -= 25
|
|
return score
|
|
}
|
|
|
|
async function collectDocxFiles(dir) {
|
|
const out = []
|
|
const items = await fs.readdir(dir, { withFileTypes: true })
|
|
for (const item of items) {
|
|
const fullPath = path.join(dir, item.name)
|
|
if (item.isDirectory()) {
|
|
out.push(...await collectDocxFiles(fullPath))
|
|
continue
|
|
}
|
|
if (!item.isFile()) continue
|
|
if (!item.name.toLowerCase().endsWith('.docx')) continue
|
|
if (item.name.startsWith('~$')) continue
|
|
out.push(fullPath)
|
|
}
|
|
return out
|
|
}
|
|
|
|
function pickBestPerEpisode(docxFiles) {
|
|
const byEpisode = new Map()
|
|
|
|
for (const filePath of docxFiles) {
|
|
const episode = parseEpisodeNumber(filePath)
|
|
if (!episode) continue
|
|
|
|
const current = byEpisode.get(episode)
|
|
const next = {
|
|
filePath,
|
|
episode,
|
|
rank: getVariantRank(filePath),
|
|
}
|
|
|
|
if (!current || next.rank > current.rank) {
|
|
byEpisode.set(episode, next)
|
|
}
|
|
}
|
|
|
|
return [...byEpisode.values()].sort((a, b) => a.episode - b.episode)
|
|
}
|
|
|
|
function extractDocText(filePath) {
|
|
const output = execFileSync('textutil', ['-convert', 'txt', '-stdout', filePath], { encoding: 'utf8' })
|
|
return output
|
|
}
|
|
|
|
function normalizeContent(text) {
|
|
const lines = text
|
|
.split(/\r?\n/)
|
|
.map(line => line.replace(/\s+/g, ' ').trim())
|
|
.filter(Boolean)
|
|
|
|
const filtered = lines.filter(line => {
|
|
const upper = line.toUpperCase()
|
|
if (upper === 'VERSE BY VERSE WITH NATE') return false
|
|
if (upper === 'A JOURNEY THROUGH SCRIPTURE') return false
|
|
return true
|
|
})
|
|
|
|
return filtered.join(' ').replace(/\s{2,}/g, ' ').trim()
|
|
}
|
|
|
|
function buildKeywords(title, content, existingKeywords = []) {
|
|
const tokens = `${title} ${content.slice(0, 1600)}`
|
|
.toLowerCase()
|
|
.replace(/[^a-z0-9\s:-]/g, ' ')
|
|
.split(/\s+/)
|
|
.filter(token => token.length >= 3 && !STOP_WORDS.has(token))
|
|
|
|
const counts = new Map()
|
|
for (const token of tokens) {
|
|
counts.set(token, (counts.get(token) ?? 0) + 1)
|
|
}
|
|
|
|
const top = [...counts.entries()]
|
|
.sort((a, b) => b[1] - a[1])
|
|
.slice(0, 20)
|
|
.map(([token]) => token)
|
|
|
|
return [...new Set([...(existingKeywords ?? []), ...top])].slice(0, 25)
|
|
}
|
|
|
|
function getEpisodeFromTitle(title = '') {
|
|
const match = title.match(/Episode\s+(\d+)/i)
|
|
return match ? Number(match[1]) : null
|
|
}
|
|
|
|
function getEntryTitleFallback(episodeNumber, rawText, existingTitle) {
|
|
if (existingTitle && existingTitle.trim()) return existingTitle
|
|
|
|
const lineMatch = rawText.match(new RegExp(`EPISODE\\s+${episodeNumber}\\s*[—-]\\s*([^\\n]+)`, 'i'))
|
|
if (lineMatch) {
|
|
return `Episode ${episodeNumber} — ${lineMatch[1].trim()}`
|
|
}
|
|
return `Episode ${episodeNumber}`
|
|
}
|
|
|
|
async function run() {
|
|
const raw = await fs.readFile(CHATBOT_FILE, 'utf8')
|
|
const entries = JSON.parse(raw)
|
|
|
|
const docxFiles = await collectDocxFiles(DOCS_DIR)
|
|
const selected = pickBestPerEpisode(docxFiles)
|
|
|
|
const existingByEpisode = new Map()
|
|
for (const entry of entries) {
|
|
const episode = getEpisodeFromTitle(entry.title)
|
|
if (episode) existingByEpisode.set(episode, entry)
|
|
}
|
|
|
|
const now = new Date().toISOString()
|
|
let updated = 0
|
|
let added = 0
|
|
|
|
for (const item of selected) {
|
|
const rawText = extractDocText(item.filePath)
|
|
const content = normalizeContent(rawText)
|
|
if (!content) continue
|
|
|
|
const existing = existingByEpisode.get(item.episode)
|
|
|
|
if (existing) {
|
|
existing.type = 'episode'
|
|
existing.title = getEntryTitleFallback(item.episode, rawText, existing.title)
|
|
existing.content = content
|
|
existing.keywords = buildKeywords(existing.title, content, existing.keywords)
|
|
existing.updatedAt = now
|
|
updated += 1
|
|
continue
|
|
}
|
|
|
|
entries.push({
|
|
id: randomUUID(),
|
|
type: 'episode',
|
|
title: getEntryTitleFallback(item.episode, rawText, ''),
|
|
content,
|
|
keywords: buildKeywords(`Episode ${item.episode}`, content, []),
|
|
createdAt: now,
|
|
updatedAt: now,
|
|
})
|
|
added += 1
|
|
}
|
|
|
|
entries.sort((a, b) => {
|
|
const aEp = getEpisodeFromTitle(a.title)
|
|
const bEp = getEpisodeFromTitle(b.title)
|
|
if (aEp && bEp) return aEp - bEp
|
|
if (aEp && !bEp) return 1
|
|
if (!aEp && bEp) return -1
|
|
return 0
|
|
})
|
|
|
|
await fs.writeFile(CHATBOT_FILE, `${JSON.stringify(entries, null, 2)}\n`)
|
|
|
|
console.log(`Episodes selected from docs: ${selected.length}`)
|
|
console.log(`Updated entries: ${updated}`)
|
|
console.log(`Added entries: ${added}`)
|
|
for (const item of selected) {
|
|
console.log(`- Episode ${item.episode}: ${path.relative(ROOT, item.filePath)}`)
|
|
}
|
|
}
|
|
|
|
run().catch(error => {
|
|
console.error(error)
|
|
process.exitCode = 1
|
|
})
|