From e0a10aae3e9cd1c7c177680d045e649b8482f7a4 Mon Sep 17 00:00:00 2001 From: harshit-epyc Date: Fri, 14 Aug 2026 18:35:46 +0530 Subject: [PATCH 01/12] feat(tools): AI chatbot diagnostic + embeddable widget Adds /tools/ai-chatbot: a visitor pastes a URL, we crawl up to 20 pages, they chat with a bot built from that corpus, and we show a scored report on what their site cannot answer. Verified email holders can then claim an embed key and run the same bot on their own site. Crawler (lib/crawl/, shared with the planned Website Grader and llms.txt Generator): - URL safety gate; every redirect hop re-validated, not just the seed - robots.txt parsing, sitemap discovery via the Sitemap: directive, and one level of sitemap index (most real sitemaps are indexes) - HTML to text with heading metadata in a single pass - caps: 20 pages, 500KB/page, 20s wall clock, 5 concurrent, 3 redirects, identified User-Agent Platform (lib/tools/): - atomic daily counters, corrected from the spec's UPDATE form which reported "capped" on the first request of every day - OpenRouter client with a model chain; reasoning disabled after it leaked chain-of-thought into answers and made replies 7x slower - site-checks.ts: structure, crawlability and specificity, tool-agnostic Chatbot (lib/tools/chatbot/): - system prompt that answers only from the crawled text and says so plainly when it cannot - report: three measured scores at crawl time, answerability and coverage from one background model call, evidence required per question - embed keys bound to the crawled host, per-key daily cap, attribution - email verification: HMAC'd codes, 10 min expiry, 5 attempts, single use, 3 per email per day, 3 per session. Sending is stubbed until a provider exists; codes are logged Also: CI ran on branch names this repo no longer uses, so it never ran on a pull request. Fixed, moved to Node 22, and given a test step. 75 tests. Known gaps are tracked in docs/ai-chatbot-plan.md. The route layer has no automated tests, nothing prunes stored page text yet, and the privacy line is unwritten. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/ci.yml | 12 +- .gitignore | 3 + app/(my-app)/tools/ai-chatbot/page.tsx | 54 + app/api/embed/chatbot.js/route.ts | 208 ++++ app/api/embed/chatbot/message/route.ts | 142 +++ app/api/tools/chatbot/crawl/route.ts | 180 ++++ app/api/tools/chatbot/diagnosis/route.ts | 95 ++ app/api/tools/chatbot/embed/route.ts | 97 ++ app/api/tools/chatbot/message/route.ts | 167 ++++ app/api/tools/chatbot/verify/route.ts | 94 ++ cloudflare-env.secrets.d.ts | 44 + components/sections/chatbot-tool.tsx | 1109 +++++++++++++++++++++ data/buyer-questions.ts | 88 ++ db/migrations/0003_tool_sessions.sql | 60 ++ db/migrations/0004_tool_embeds.sql | 34 + db/migrations/0005_tool_verifications.sql | 29 + docs/ai-chatbot-architecture.md | 179 ++++ docs/ai-chatbot-flow.md | 453 +++++++++ docs/ai-chatbot-plan.md | 130 +++ docs/ai-chatbot-tech.md | 443 ++++++++ lib/crawl/extract.test.ts | 98 ++ lib/crawl/extract.ts | 131 +++ lib/crawl/fetch-pages.test.ts | 181 ++++ lib/crawl/fetch-pages.ts | 298 ++++++ lib/crawl/sitemap.test.ts | 156 +++ lib/crawl/sitemap.ts | 175 ++++ lib/crawl/validate-url.test.ts | 96 ++ lib/crawl/validate-url.ts | 78 ++ lib/image-loader.ts | 13 +- lib/tools/chatbot/diagnosis.ts | 166 +++ lib/tools/chatbot/embed.ts | 136 +++ lib/tools/chatbot/prompt.ts | 85 ++ lib/tools/chatbot/schema.ts | 56 ++ lib/tools/chatbot/verification.ts | 188 ++++ lib/tools/counters.test.ts | 122 +++ lib/tools/counters.ts | 105 ++ lib/tools/email.ts | 99 ++ lib/tools/models.ts | 215 ++++ lib/tools/session.ts | 245 +++++ lib/tools/site-checks.test.ts | 143 +++ lib/tools/site-checks.ts | 260 +++++ lib/tools/sse-client.ts | 45 + package.json | 4 +- pnpm-lock.yaml | 663 +++++++++++- vitest.config.mts | 18 + 45 files changed, 7350 insertions(+), 47 deletions(-) create mode 100644 app/(my-app)/tools/ai-chatbot/page.tsx create mode 100644 app/api/embed/chatbot.js/route.ts create mode 100644 app/api/embed/chatbot/message/route.ts create mode 100644 app/api/tools/chatbot/crawl/route.ts create mode 100644 app/api/tools/chatbot/diagnosis/route.ts create mode 100644 app/api/tools/chatbot/embed/route.ts create mode 100644 app/api/tools/chatbot/message/route.ts create mode 100644 app/api/tools/chatbot/verify/route.ts create mode 100644 cloudflare-env.secrets.d.ts create mode 100644 components/sections/chatbot-tool.tsx create mode 100644 data/buyer-questions.ts create mode 100644 db/migrations/0003_tool_sessions.sql create mode 100644 db/migrations/0004_tool_embeds.sql create mode 100644 db/migrations/0005_tool_verifications.sql create mode 100644 docs/ai-chatbot-architecture.md create mode 100644 docs/ai-chatbot-flow.md create mode 100644 docs/ai-chatbot-plan.md create mode 100644 docs/ai-chatbot-tech.md create mode 100644 lib/crawl/extract.test.ts create mode 100644 lib/crawl/extract.ts create mode 100644 lib/crawl/fetch-pages.test.ts create mode 100644 lib/crawl/fetch-pages.ts create mode 100644 lib/crawl/sitemap.test.ts create mode 100644 lib/crawl/sitemap.ts create mode 100644 lib/crawl/validate-url.test.ts create mode 100644 lib/crawl/validate-url.ts create mode 100644 lib/tools/chatbot/diagnosis.ts create mode 100644 lib/tools/chatbot/embed.ts create mode 100644 lib/tools/chatbot/prompt.ts create mode 100644 lib/tools/chatbot/schema.ts create mode 100644 lib/tools/chatbot/verification.ts create mode 100644 lib/tools/counters.test.ts create mode 100644 lib/tools/counters.ts create mode 100644 lib/tools/email.ts create mode 100644 lib/tools/models.ts create mode 100644 lib/tools/session.ts create mode 100644 lib/tools/site-checks.test.ts create mode 100644 lib/tools/site-checks.ts create mode 100644 lib/tools/sse-client.ts create mode 100644 vitest.config.mts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0b02598..08aa44a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,10 @@ name: CI on: pull_request: - branches: [develop, master, production] + # `main` is the default branch and `production` is the release branch. + # These were previously [develop, master, production] — branch names this + # repo no longer uses — so CI never ran on a normal pull request. + branches: [main, production] workflow_dispatch: jobs: @@ -19,7 +22,9 @@ jobs: - uses: actions/setup-node@v4 with: - node-version: 20 + # 22, not 20: lib/tools/counters.test.ts runs its SQL against the + # built-in `node:sqlite`, which does not exist before Node 22. + node-version: 22 cache: pnpm - run: pnpm install --frozen-lockfile @@ -27,6 +32,9 @@ jobs: - name: Typecheck run: pnpm exec tsc --noEmit + - name: Test + run: pnpm test + - name: Typecheck contact-webhook worker # The consumer Worker is excluded from the root tsconfig (different # libs), so it has its own project — check it explicitly. diff --git a/.gitignore b/.gitignore index 8f56b6d..7e5dc36 100644 --- a/.gitignore +++ b/.gitignore @@ -61,3 +61,6 @@ graphify-out/ /.design-sync/.cache/ /.design-sync/learnings/ /.design-sync/node_modules + +# Local-only embed widget test page — never ship this +public/embed-test.html diff --git a/app/(my-app)/tools/ai-chatbot/page.tsx b/app/(my-app)/tools/ai-chatbot/page.tsx new file mode 100644 index 0000000..272338b --- /dev/null +++ b/app/(my-app)/tools/ai-chatbot/page.tsx @@ -0,0 +1,54 @@ +import type { Metadata } from 'next' +import { SiteNav } from '@/components/site-nav' +import { ChatbotTool } from '@/components/sections/chatbot-tool' +import { CTAFooter } from '@/components/sections/cta-footer' +import { Button } from '@/components/ui/button' +import { Container } from '@/components/ui/container' +import { Section } from '@/components/ui/section' + +/** + * WIREFRAME — not wired to anything yet. + * + * Plan: docs/ai-chatbot-plan.md · Architecture: docs/ai-chatbot-architecture.md + * Build reference: docs/ai-chatbot-tech.md + * + * Left out of the sitemap deliberately until the tool actually works — see + * step 8 (Launch) in the plan. + */ +export const metadata: Metadata = { + title: 'Can an AI read your website?', + description: + 'Paste your website address. We read up to 20 pages, build a chatbot from what we find, and show you which of the 10 questions a buyer asks your site cannot answer.', + alternates: { canonical: '/tools/ai-chatbot' }, + // Wireframe — keep it out of the index until the tool is real. + robots: { index: false, follow: false }, +} + +export default function AiChatbotToolPage() { + return ( + <> +
+ + + +
+ + + + + + + + } + /> + + ) +} diff --git a/app/api/embed/chatbot.js/route.ts b/app/api/embed/chatbot.js/route.ts new file mode 100644 index 0000000..be0b51a --- /dev/null +++ b/app/api/embed/chatbot.js/route.ts @@ -0,0 +1,208 @@ +/** + * The embeddable widget, served as one self-contained script. + * + * Hand-written, no framework, no external fonts, no CDN. This runs on someone + * else's production site, so its weight is our craft on display — a studio + * selling performance cannot ship a heavy widget. + * + * Rendered into a shadow root rather than an iframe. The iframe the spec + * proposed would isolate CSS, but it would also make every message request + * originate from our domain, which breaks the key's domain binding — see + * app/api/embed/chatbot/message/route.ts. A shadow root isolates styles in + * both directions and keeps the request genuinely cross-origin. + * + * Cached hard at the edge: this changes rarely and sits on the critical path + * of other people's pages. + */ + +const WIDGET = String.raw` +(function () { + var script = document.currentScript; + if (!script) return; + var key = script.getAttribute('data-key'); + if (!key) return; + + var api = new URL(script.src).origin; + var mounted = false; + + function el(tag, css, text) { + var node = document.createElement(tag); + if (css) node.setAttribute('style', css); + if (text) node.textContent = text; + return node; + } + + var INK = '#183228'; + var CRIMSON = '#b91646'; + var CREAM = '#fff0d0'; + var FONT = '-apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif'; + + var host = el('div'); + host.setAttribute('data-epyc-chat', ''); + var root = host.attachShadow({ mode: 'open' }); + + var wrap = el('div', 'position:fixed;right:20px;bottom:20px;z-index:2147483000;font-family:' + FONT + ';'); + + var launcher = el('button', [ + 'width:56px;height:56px;border-radius:50%;border:0;cursor:pointer', + 'background:' + CRIMSON, 'color:' + CREAM, 'font-size:22px;line-height:1', + 'box-shadow:0 6px 20px rgba(0,0,0,.22)' + ].join(';'), '✧'); + launcher.setAttribute('aria-label', 'Open chat'); + + var panel = el('div', [ + 'display:none;flex-direction:column;width:360px;max-width:calc(100vw - 40px)', + 'height:520px;max-height:calc(100vh - 120px);margin-bottom:12px', + 'background:#fff;border-radius:16px;overflow:hidden', + 'box-shadow:0 12px 40px rgba(0,0,0,.24)' + ].join(';')); + + var header = el('div', 'padding:14px 16px;background:' + INK + ';color:' + CREAM + ';font-size:14px;font-weight:600;'); + var headerText = el('span', null, 'Ask us anything'); + header.appendChild(headerText); + + var log = el('div', 'flex:1;overflow-y:auto;padding:16px;display:flex;flex-direction:column;gap:10px;background:#faf8f3;'); + + var form = el('form', 'display:flex;gap:8px;padding:12px;border-top:1px solid #e8e3d8;background:#fff;'); + var input = el('input', 'flex:1;min-width:0;padding:10px 12px;border:1px solid #ddd6c7;border-radius:10px;font:inherit;font-size:14px;outline:none;'); + input.setAttribute('placeholder', 'Type your question'); + input.setAttribute('aria-label', 'Your question'); + var send = el('button', 'padding:10px 14px;border:0;border-radius:10px;background:' + CRIMSON + ';color:' + CREAM + ';cursor:pointer;font:inherit;font-size:14px;', 'Send'); + send.type = 'submit'; + form.appendChild(input); + form.appendChild(send); + + var footer = el('div', 'padding:8px 12px;text-align:center;font-size:11px;color:#8a8577;background:#fff;'); + var credit = el('a', 'color:#8a8577;text-decoration:none;', 'Powered by EPYC'); + credit.href = api + '/tools/ai-chatbot?utm_source=embed&utm_medium=widget'; + credit.target = '_blank'; + credit.rel = 'noopener'; + footer.appendChild(credit); + + panel.appendChild(header); + panel.appendChild(log); + panel.appendChild(form); + panel.appendChild(footer); + wrap.appendChild(panel); + wrap.appendChild(launcher); + root.appendChild(wrap); + + function bubble(who, text) { + var mine = who === 'you'; + var b = el('div', [ + 'max-width:82%;padding:9px 12px;border-radius:12px;font-size:14px;line-height:1.45', + 'white-space:pre-wrap;word-wrap:break-word', + mine ? 'align-self:flex-end;background:' + INK + ';color:' + CREAM + : 'align-self:flex-start;background:#efe9dc;color:#1c1c1c' + ].join(';'), text); + log.appendChild(b); + log.scrollTop = log.scrollHeight; + return b; + } + + var history = []; + var busy = false; + + function ask(text) { + if (busy || !text) return; + busy = true; + input.value = ''; + bubble('you', text); + var out = bubble('bot', '...'); + var answer = ''; + + fetch(api + '/api/embed/chatbot/message', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ key: key, message: text, history: history.slice(-8) }) + }).then(function (res) { + if (!res.ok || !res.body) { + return res.json().catch(function () { return {}; }).then(function (body) { + out.textContent = body.error || 'Something went wrong. Please try again.'; + busy = false; + }); + } + + var reader = res.body.getReader(); + var decoder = new TextDecoder(); + var buffer = ''; + + function pump() { + return reader.read().then(function (chunk) { + if (chunk.done) { + history.push({ role: 'user', content: text }); + history.push({ role: 'assistant', content: answer }); + busy = false; + return; + } + buffer += decoder.decode(chunk.value, { stream: true }); + var split; + while ((split = buffer.indexOf('\n\n')) !== -1) { + var frame = buffer.slice(0, split); + buffer = buffer.slice(split + 2); + var data = ''; + frame.split('\n').forEach(function (line) { + if (line.indexOf('data:') === 0) data += line.slice(5).trim(); + }); + if (!data) continue; + try { + var parsed = JSON.parse(data); + if (parsed.text) { + answer += parsed.text; + out.textContent = answer; + log.scrollTop = log.scrollHeight; + } else if (parsed.message) { + out.textContent = answer + '\n\n' + parsed.message; + } + } catch (e) { /* keep-alive or partial frame */ } + } + return pump(); + }); + } + return pump(); + }).catch(function () { + out.textContent = 'I could not reach the assistant. Please try again.'; + busy = false; + }); + } + + form.addEventListener('submit', function (e) { + e.preventDefault(); + ask(input.value.trim()); + }); + + launcher.addEventListener('click', function () { + var open = panel.style.display === 'flex'; + panel.style.display = open ? 'none' : 'flex'; + launcher.textContent = open ? '✧' : '×'; + launcher.setAttribute('aria-label', open ? 'Open chat' : 'Close chat'); + if (!open) { + input.focus(); + if (!mounted) { + mounted = true; + bubble('bot', 'Hi — ask me anything about this site.'); + } + } + }); + + function mount() { + if (document.body) document.body.appendChild(host); + } + if (document.readyState === 'loading') { + document.addEventListener('DOMContentLoaded', mount); + } else { + mount(); + } +})(); +` + +export async function GET() { + return new Response(WIDGET, { + headers: { + 'content-type': 'application/javascript; charset=utf-8', + // Long cache: this sits on the critical path of other people's pages. + 'cache-control': 'public, max-age=3600, s-maxage=86400', + 'access-control-allow-origin': '*', + }, + }) +} diff --git a/app/api/embed/chatbot/message/route.ts b/app/api/embed/chatbot/message/route.ts new file mode 100644 index 0000000..e3802d1 --- /dev/null +++ b/app/api/embed/chatbot/message/route.ts @@ -0,0 +1,142 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { + consumeEmbedMessage, + corsHeaders, + findEmbedByKey, + isEmbedKey, + originAllowed, + touchEmbed, +} from '@/lib/tools/chatbot/embed' +import { embedMessageSchema } from '@/lib/tools/chatbot/schema' +import { buildMessages } from '@/lib/tools/chatbot/prompt' +import { streamChat } from '@/lib/tools/models' +import { loadPages } from '@/lib/tools/session' + +/** + * A message to a live widget running on a customer's own site. + * + * Called cross-origin from their page, which is deliberate: it means the + * `Origin` header is genuinely theirs and the key's domain binding actually + * means something. An iframe we host would report our own origin instead. + * + * Binding is not airtight — a non-browser client can send any Origin it likes. + * The per-key daily cap is what bounds that: the worst case is a fixed amount + * of free-tier inference, not an open tap. + */ + +export async function OPTIONS(req: Request) { + const origin = req.headers.get('origin') + if (!origin) return new Response(null, { status: 403 }) + + // The preflight cannot see the key, so it is answered permissively for the + // requesting origin. The POST below is where binding is actually enforced. + return new Response(null, { status: 204, headers: corsHeaders(origin) }) +} + +export async function POST(req: Request) { + const origin = req.headers.get('origin') + + const json = await req.json().catch(() => null) + const parsed = embedMessageSchema.safeParse(json) + if (!parsed.success || !isEmbedKey(parsed.data.key)) { + return NextResponse.json({ ok: false, error: 'Invalid request.' }, { status: 400 }) + } + + const { env } = getCloudflareContext() + const db = env.DB + + const embed = await findEmbedByKey(db, parsed.data.key) + if (!embed || embed.status !== 'active') { + return NextResponse.json({ ok: false, error: 'This assistant is not available.' }, { status: 404 }) + } + + // The binding. A key minted for acme.com answers only for acme.com. + if (!originAllowed(origin, embed.bound_host, { + allowAny: env.TOOLS_EMBED_ALLOW_ANY_ORIGIN === 'true', + })) { + return NextResponse.json( + { ok: false, error: 'This assistant is not available on this domain.' }, + { status: 403 }, + ) + } + + const cors = corsHeaders(origin!) + + const apiKey = env.OPENROUTER_API_KEY + if (!apiKey) { + console.error('OPENROUTER_API_KEY is not set') + return NextResponse.json( + { ok: false, error: 'The assistant is unavailable.' }, + { status: 503, headers: cors }, + ) + } + + if (!(await consumeEmbedMessage(db, embed.key))) { + return NextResponse.json( + { ok: false, error: 'This assistant has reached its limit for today.' }, + { status: 429, headers: cors }, + ) + } + + const pages = await loadPages(db, embed.session_id) + const messages = buildMessages( + embed.bound_host, + pages, + parsed.data.history, + parsed.data.message, + ) + + let result + try { + result = await streamChat({ + apiKey, + messages, + allowPaid: env.OPENROUTER_ALLOW_PAID === 'true', + }) + } catch (err) { + console.error('embed: all model tiers unavailable', err) + return NextResponse.json( + { ok: false, error: 'I’m having trouble answering right now. Please try again shortly.' }, + { status: 503, headers: cors }, + ) + } + + await touchEmbed(db, embed.key).catch(() => {}) + + const encoder = new TextEncoder() + const stream = new ReadableStream({ + async start(controller) { + const send = (event: string, data: unknown) => + controller.enqueue(encoder.encode(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`)) + + try { + const reader = result.stream.getReader() + for (;;) { + const { done, value } = await reader.read() + if (done) break + if (value) send('delta', { text: value }) + } + } catch (err) { + console.error('embed: stream failed mid-answer', err) + send('error', { message: 'That answer was cut short.' }) + } finally { + send('done', {}) + try { + controller.close() + } catch { + // Already closed. + } + } + }, + }) + + return new Response(stream, { + headers: { + ...cors, + 'content-type': 'text/event-stream; charset=utf-8', + 'cache-control': 'no-cache, no-transform', + 'x-accel-buffering': 'no', + }, + }) +} diff --git a/app/api/tools/chatbot/crawl/route.ts b/app/api/tools/chatbot/crawl/route.ts new file mode 100644 index 0000000..f30414d --- /dev/null +++ b/app/api/tools/chatbot/crawl/route.ts @@ -0,0 +1,180 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { crawlSite, type CrawlProgress } from '@/lib/crawl/fetch-pages' +import { validateUrl } from '@/lib/crawl/validate-url' +import { crawlSchema } from '@/lib/tools/chatbot/schema' +import { bumpCounter, capsFor, counterKeys, underLimit } from '@/lib/tools/counters' +import { scoreDeterministic } from '@/lib/tools/chatbot/diagnosis' +import { + copyPages, + createSession, + finishSession, + findRecentCrawl, + getSession, + hashIp, + loadPagesForScoring, + savePages, + saveDiagnosis, +} from '@/lib/tools/session' + +/** + * Read a visitor's website and build the corpus their chatbot answers from. + * + * Runs the crawl inline and streams progress as Server-Sent Events, rather + * than handing it to a queue and polling for status — see + * docs/ai-chatbot-architecture.md §2.2. The whole crawl is capped at 20 + * seconds, so this is a short-lived request, not a long-poll. + * + * Every event is one line of `event:` + `data:`. The client renders `page` + * events as the live log and waits for `done`. + */ + +export async function POST(req: Request) { + const json = await req.json().catch(() => null) + const parsed = crawlSchema.safeParse(json) + if (!parsed.success) { + return NextResponse.json({ ok: false, error: 'Enter a website address.' }, { status: 400 }) + } + + // Cheapest rejection first: a bad address never reaches the network. + const checked = validateUrl(parsed.data.url) + if (!checked.ok) { + return NextResponse.json({ ok: false, error: checked.reason }, { status: 400 }) + } + + const { env } = getCloudflareContext() + const db = env.DB + + const ip = req.headers.get('cf-connecting-ip') ?? '0.0.0.0' + const ipHash = await hashIp(ip, env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production') + const ipKey = counterKeys.ip(ipHash) + + // Checked, not consumed: a crawl that fails should not cost the visitor one + // of their three. The counter is bumped once a session actually exists. + const caps = capsFor(env) + if (!(await underLimit(db, ipKey, caps.sessionsPerIp))) { + return NextResponse.json( + { + ok: false, + error: `You've used your ${caps.sessionsPerIp} checks for today. They reset at midnight UTC.`, + capped: true, + }, + { status: 429 }, + ) + } + + const sessionId = crypto.randomUUID() + const encoder = new TextEncoder() + + const stream = new ReadableStream({ + async start(controller) { + const send = (event: string, data: unknown) => { + controller.enqueue(encoder.encode(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`)) + } + + try { + await createSession(db, { + id: sessionId, + tool: 'chatbot', + targetUrl: checked.url, + host: checked.host, + ipHash, + status: 'crawling', + }) + + // A recent read of the same host is copied instead of re-fetched. + if (!parsed.data.force) { + const recent = await findRecentCrawl(db, checked.host) + if (recent) { + const copied = await copyPages(db, recent.id, sessionId) + await finishSession(db, sessionId, 'ready', copied) + await bumpCounter(db, ipKey, caps.sessionsPerIp) + + // Copy the report too, not just the pages. Same host, same corpus, + // so the scores are identical — and it saves a model call. + // Without this the session has no diagnosis at all, and the later + // background merge produces an object with answerability but no + // structure or crawlability, which the report then dereferences. + const source = await getSession(db, recent.id) + if (source?.diagnosis_json) { + await saveDiagnosis(db, sessionId, JSON.parse(source.diagnosis_json)) + } else { + const pages = await loadPagesForScoring(db, sessionId) + await saveDiagnosis(db, sessionId, scoreDeterministic(pages, {}, false)) + } + send('status', { message: 'We read this site recently — using that.' }) + send('done', { sessionId, pages: copied, status: 'ready', reused: true }) + // No close() here — `return` runs the finally block below, which + // closes exactly once. Closing here too threw + // "Invalid state: Controller is already closed" and failed the + // whole response, on the one path that was supposed to be fastest. + return + } + } + + const result = await crawlSite(checked.url, { + onProgress: (p: CrawlProgress) => { + if (p.type === 'status') send('status', { message: p.message }) + else send('page', { url: pathOf(p.url), title: p.title, done: p.done, total: p.total }) + }, + }) + + const readable = result.pages.filter((p) => !p.isEmpty) + const status = result.signals.robotsBlockedAll || result.signals.unreachable + ? 'failed' + : readable.length === 0 + ? 'empty' + : 'ready' + + await savePages(db, sessionId, result.pages) + await finishSession(db, sessionId, status, result.pages.length) + await bumpCounter(db, ipKey, caps.sessionsPerIp) + + // Three of the five dimensions are free — they come straight from what + // we just extracted. Scoring them now means a visitor who never sends a + // message still has most of a report, and the empty-site path (which + // has no message one to trigger on) is covered. + await saveDiagnosis(db, sessionId, scoreDeterministic(result.pages, result.signals)) + + send('done', { + sessionId, + pages: result.pages.length, + readablePages: readable.length, + status, + signals: result.signals, + }) + } catch (err) { + console.error('crawl failed', err) + await finishSession(db, sessionId, 'failed', 0).catch(() => {}) + send('error', { message: "We couldn't read that site. Try another address." }) + } finally { + // Closing an already-closed controller throws and fails the response, + // so this is the single close for every path through the handler. + try { + controller.close() + } catch { + // Already closed — nothing to do. + } + } + }, + }) + + // Keep the connection un-buffered end to end. `x-accel-buffering` is ignored + // by Cloudflare but matters behind any proxy in front of a local dev server. + return new Response(stream, { + headers: { + 'content-type': 'text/event-stream; charset=utf-8', + 'cache-control': 'no-cache, no-transform', + 'x-accel-buffering': 'no', + }, + }) +} + +/** The log shows paths, not full URLs — shorter, and it reads as *their* site. */ +function pathOf(url: string): string { + try { + return new URL(url).pathname + } catch { + return url + } +} diff --git a/app/api/tools/chatbot/diagnosis/route.ts b/app/api/tools/chatbot/diagnosis/route.ts new file mode 100644 index 0000000..d95e15e --- /dev/null +++ b/app/api/tools/chatbot/diagnosis/route.ts @@ -0,0 +1,95 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { + mergeJudged, + scoreDeterministic, + scoreWithModel, + type Diagnosis, +} from '@/lib/tools/chatbot/diagnosis' +import { + getSession, + loadPages, + loadPagesForScoring, + saveDiagnosis, +} from '@/lib/tools/session' + +/** + * The report for a session. + * + * Normally this is a read: the deterministic three were stored when the crawl + * finished, and the model call filled in the other two in the background on + * message one. If neither has happened, it computes the deterministic three + * inline rather than returning an error — a partial report converts, an error + * state does not. + */ +export async function GET(req: Request) { + const sessionId = new URL(req.url).searchParams.get('sessionId') + if (!sessionId) { + return NextResponse.json({ ok: false, error: 'Missing session.' }, { status: 400 }) + } + + const { env, ctx } = getCloudflareContext() + const db = env.DB + + const session = await getSession(db, sessionId) + if (!session) { + return NextResponse.json({ ok: false, error: 'That session has expired.' }, { status: 404 }) + } + + if (session.diagnosis_json) { + try { + let stored = JSON.parse(session.diagnosis_json) as Diagnosis + + // Repair rather than serve a broken report. A stored diagnosis can be + // missing its measured half — sessions written before the reuse path + // stored one, or any future write that skips it — and rendering that + // takes the report down. Rebuild the measured three from the pages we + // still have, keep whatever judged half exists, and write it back so the + // repair happens once rather than on every read. + if (!stored?.structure || !stored.crawlability || !stored.specificity) { + const pages = await loadPagesForScoring(db, sessionId) + stored = { + ...scoreDeterministic(pages, {}, false), + answerability: stored?.answerability ?? null, + coverage: stored?.coverage ?? null, + partial: !stored?.answerability, + } + await saveDiagnosis(db, sessionId, stored) + } + + // Normally the judged half is scored in the background on message one. + // A visitor who skips straight to the report never sends one, so nothing + // would ever trigger it and the panel would poll forever. Asking for the + // report is itself the signal that someone wants it, so start the call + // here and let the client's polling pick it up. + if (!stored.answerability && session.status === 'ready' && env.OPENROUTER_API_KEY) { + ctx.waitUntil( + (async () => { + try { + const pages = await loadPages(db, sessionId) + const judged = await scoreWithModel(env.OPENROUTER_API_KEY!, session.host, pages, { + allowPaid: env.OPENROUTER_ALLOW_PAID === 'true', + }) + const scoringPages = await loadPagesForScoring(db, sessionId) + await saveDiagnosis(db, sessionId, mergeJudged(stored, judged, scoringPages)) + } catch (err) { + console.error('diagnosis scoring failed', err) + } + })(), + ) + } + + return NextResponse.json({ ok: true, host: session.host, diagnosis: stored }) + } catch { + // Fall through and recompute rather than failing on a bad row. + } + } + + // No stored diagnosis means we never saw this session's crawl, so the + // sitemap and robots findings are unknown — `false` here omits them rather + // than reporting an absence we cannot vouch for. + const pages = await loadPagesForScoring(db, sessionId) + const diagnosis = scoreDeterministic(pages, {}, false) + + return NextResponse.json({ ok: true, host: session.host, diagnosis }) +} diff --git a/app/api/tools/chatbot/embed/route.ts b/app/api/tools/chatbot/embed/route.ts new file mode 100644 index 0000000..472dd64 --- /dev/null +++ b/app/api/tools/chatbot/embed/route.ts @@ -0,0 +1,97 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { claimSchema } from '@/lib/tools/chatbot/schema' +import { + createEmbed, + embedSnippet, + findEmbedBySession, + mintKey, +} from '@/lib/tools/chatbot/embed' +import { isVerified } from '@/lib/tools/chatbot/verification' +import { getSession } from '@/lib/tools/session' + +/** + * Claim an embed: mint a key for a verified address and return the snippet. + * + * The address must already have completed the code exchange — see + * app/api/tools/chatbot/verify/route.ts. Delivery of that code is stubbed + * until a provider exists, but the exchange itself is real, so this gate holds + * either way. + * + * Claiming makes this session's corpus permanent: the live bot answers from + * it, so any pruning job must skip sessions with an embed. + */ +export async function POST(req: Request) { + const json = await req.json().catch(() => null) + const parsed = claimSchema.safeParse(json) + if (!parsed.success) { + return NextResponse.json( + { ok: false, error: 'Enter a valid email address.' }, + { status: 400 }, + ) + } + + const { env } = getCloudflareContext() + const db = env.DB + + const session = await getSession(db, parsed.data.sessionId) + if (!session) { + return NextResponse.json({ ok: false, error: 'That session has expired.' }, { status: 404 }) + } + + // A bot with nothing to answer from would embarrass whoever installs it. + if (session.status !== 'ready') { + return NextResponse.json( + { ok: false, error: 'There is not enough on that site to build a bot from yet.' }, + { status: 409 }, + ) + } + + // The gate. A key is only minted for an address that proved it owns the + // inbox — otherwise the email is worth no more than the click that preceded it. + const verified = await isVerified(db, session.id, parsed.data.email) + if (!verified) { + return NextResponse.json( + { ok: false, error: 'Verify your email address first.', needsVerification: true }, + { status: 403 }, + ) + } + + const origin = new URL(req.url).origin + + // Claiming twice returns the same key rather than minting a second one — + // otherwise a refresh silently orphans the snippet they already pasted. + const existing = await findEmbedBySession(db, session.id) + if (existing) { + return NextResponse.json({ + ok: true, + key: existing.key, + host: existing.bound_host, + snippet: embedSnippet(existing.key, origin), + alreadyClaimed: true, + }) + } + + const key = mintKey() + await createEmbed(db, { + key, + sessionId: session.id, + boundHost: session.host.replace(/^www\./, ''), + email: parsed.data.email, + crawledAt: new Date().toISOString(), + }) + + // Mirror the email onto the session too, so the funnel reads in one place. + await db + .prepare('UPDATE tool_sessions SET email = ? WHERE id = ?') + .bind(parsed.data.email, session.id) + .run() + + return NextResponse.json({ + ok: true, + key, + host: session.host, + snippet: embedSnippet(key, origin), + alreadyClaimed: false, + }) +} diff --git a/app/api/tools/chatbot/message/route.ts b/app/api/tools/chatbot/message/route.ts new file mode 100644 index 0000000..de87b59 --- /dev/null +++ b/app/api/tools/chatbot/message/route.ts @@ -0,0 +1,167 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { CAPS, bumpCounter, counterKeys } from '@/lib/tools/counters' +import { mergeJudged, scoreWithModel, type Diagnosis } from '@/lib/tools/chatbot/diagnosis' +import { streamChat } from '@/lib/tools/models' +import { messageSchema } from '@/lib/tools/chatbot/schema' +import { buildMessages } from '@/lib/tools/chatbot/prompt' +import { + getSession, + loadPages, + loadPagesForScoring, + readTranscript, + recordTurn, + saveDiagnosis, +} from '@/lib/tools/session' + +/** + * One chat turn, streamed. + * + * The corpus is re-sent on every message — there is no prompt caching on these + * models — which is affordable only because every tier is free. See + * docs/ai-chatbot-tech.md. + */ + +export async function POST(req: Request) { + const json = await req.json().catch(() => null) + const parsed = messageSchema.safeParse(json) + if (!parsed.success) { + return NextResponse.json({ ok: false, error: 'Ask a question first.' }, { status: 400 }) + } + + const { env, ctx } = getCloudflareContext() + const db = env.DB + + const apiKey = env.OPENROUTER_API_KEY + if (!apiKey) { + console.error('OPENROUTER_API_KEY is not set') + return NextResponse.json({ ok: false, error: 'The assistant is unavailable.' }, { status: 503 }) + } + + const session = await getSession(db, parsed.data.sessionId) + if (!session) { + return NextResponse.json({ ok: false, error: 'That session has expired.' }, { status: 404 }) + } + + // The report is the end of the conversation, not an error. + if (session.messages_used >= CAPS.messagesPerSession) { + return NextResponse.json( + { ok: false, capped: true, error: 'You’ve used all your questions.' }, + { status: 409 }, + ) + } + + if (session.status !== 'ready') { + return NextResponse.json( + { ok: false, error: 'There is not enough on that site to chat about.' }, + { status: 409 }, + ) + } + + // Global cap, consumed before any model call is made. + if (!(await bumpCounter(db, counterKeys.globalMessages(), CAPS.globalMessages))) { + return NextResponse.json( + { ok: false, capped: true, error: 'The tool is busy today. Try again tomorrow.' }, + { status: 429 }, + ) + } + + const pages = await loadPages(db, session.id) + const history = readTranscript(session) + const messages = buildMessages(session.host, pages, history, parsed.data.message) + + // Score the report quietly, on the first message, while the visitor is still + // typing. By the time they reach the cap the panel renders instantly instead + // of stalling for ten seconds at the exact moment we ask for the click. + // Gating on message one rather than on the crawl means visitors who bounce + // cost nothing. Failure is swallowed: the deterministic three are already + // stored, so a failed call downgrades the report rather than breaking it. + if (session.messages_used === 0 && !session.diagnosis_json?.includes('"answerability":{')) { + ctx.waitUntil( + (async () => { + try { + const judged = await scoreWithModel(apiKey, session.host, pages, { + allowPaid: env.OPENROUTER_ALLOW_PAID === 'true', + }) + const existing = session.diagnosis_json + ? (JSON.parse(session.diagnosis_json) as Partial) + : null + const scoringPages = await loadPagesForScoring(db, session.id) + await saveDiagnosis(db, session.id, mergeJudged(existing, judged, scoringPages)) + } catch (err) { + console.error('diagnosis scoring failed', err) + } + })(), + ) + } + + let result + try { + result = await streamChat({ + apiKey, + messages, + allowPaid: env.OPENROUTER_ALLOW_PAID === 'true', + }) + } catch (err) { + console.error('all model tiers unavailable', err) + return NextResponse.json( + { ok: false, error: 'I’m having trouble reaching the model. Try again shortly.' }, + { status: 503 }, + ) + } + + const encoder = new TextEncoder() + const stream = new ReadableStream({ + async start(controller) { + const send = (event: string, data: unknown) => + controller.enqueue(encoder.encode(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`)) + + // Which tier served this — logged so we learn how often tier 1 holds. + send('model', { model: result.model }) + + let answer = '' + try { + const reader = result.stream.getReader() + for (;;) { + const { done, value } = await reader.read() + if (done) break + if (!value) continue + answer += value + send('delta', { text: value }) + } + } catch (err) { + console.error('stream failed mid-answer', err) + send('error', { message: 'That answer was cut short. Try asking again.' }) + } + + // Record even a partial answer: the count and the transcript must agree + // with what the visitor actually saw on screen. + try { + const turns = [ + ...history, + { role: 'user' as const, content: parsed.data.message }, + { role: 'assistant' as const, content: answer }, + ] + await recordTurn(db, session.id, turns) + } catch (err) { + console.error('failed to record turn', err) + } + + const used = session.messages_used + 1 + send('done', { + messagesUsed: used, + messagesLeft: Math.max(0, CAPS.messagesPerSession - used), + capped: used >= CAPS.messagesPerSession, + }) + controller.close() + }, + }) + + return new Response(stream, { + headers: { + 'content-type': 'text/event-stream; charset=utf-8', + 'cache-control': 'no-cache, no-transform', + 'x-accel-buffering': 'no', + }, + }) +} diff --git a/app/api/tools/chatbot/verify/route.ts b/app/api/tools/chatbot/verify/route.ts new file mode 100644 index 0000000..d83e170 --- /dev/null +++ b/app/api/tools/chatbot/verify/route.ts @@ -0,0 +1,94 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { sendEmail, verificationEmail } from '@/lib/tools/email' +import { verifySendSchema, verifyCheckSchema } from '@/lib/tools/chatbot/schema' +import { + CODE_TTL_MINUTES, + checkCode, + issueCode, +} from '@/lib/tools/chatbot/verification' +import { getSession } from '@/lib/tools/session' + +/** + * Send a verification code, and check one. + * + * One route with an `action` rather than two, because the pair share every + * lookup and guard. Sending is stubbed until an email provider exists — see + * lib/tools/email.ts — but everything else is real. + */ +export async function POST(req: Request) { + const json = (await req.json().catch(() => null)) as { action?: string } | null + + const { env } = getCloudflareContext() + const db = env.DB + const pepper = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + if (json?.action === 'check') { + const parsed = verifyCheckSchema.safeParse(json) + if (!parsed.success) { + return NextResponse.json({ ok: false, error: 'Enter the 6-digit code.' }, { status: 400 }) + } + + const result = await checkCode(db, { ...parsed.data, pepper }) + if (result.ok) return NextResponse.json({ ok: true }) + + // Deliberately vague on wrong-code: saying how many attempts remain helps + // a guesser more than it helps a person who mistyped. + const message = { + 'no-code': 'That code has expired. Ask for a new one.', + expired: 'That code has expired. Ask for a new one.', + 'too-many-attempts': 'Too many attempts. Ask for a new code.', + 'wrong-code': 'That code is not right.', + }[result.reason] + + return NextResponse.json({ ok: false, error: message }, { status: 400 }) + } + + const parsed = verifySendSchema.safeParse(json) + if (!parsed.success) { + return NextResponse.json({ ok: false, error: 'Enter a valid email address.' }, { status: 400 }) + } + + const session = await getSession(db, parsed.data.sessionId) + if (!session) { + return NextResponse.json({ ok: false, error: 'That session has expired.' }, { status: 404 }) + } + + const issued = await issueCode(db, { + sessionId: session.id, + email: parsed.data.email, + pepper, + }) + + if (!issued.ok) { + const message = + issued.reason === 'email-capped' + ? 'That address has been sent too many codes today. Try again tomorrow.' + : 'Too many codes requested. Try again tomorrow.' + return NextResponse.json({ ok: false, error: message }, { status: 429 }) + } + + let stubbed = false + try { + const result = await sendEmail( + env, + verificationEmail(parsed.data.email, issued.code, session.host), + ) + stubbed = result.stubbed + } catch (err) { + // The code is already stored, so a delivery failure is recoverable by + // asking for another one — but say so rather than pretending it arrived. + console.error('verification email failed', err) + return NextResponse.json( + { ok: false, error: 'We could not send that email. Try again shortly.' }, + { status: 502 }, + ) + } + + return NextResponse.json({ + ok: true, + expiresInMinutes: CODE_TTL_MINUTES, + // Tells the UI to say where the code actually went. Never the code itself. + stubbed, + }) +} diff --git a/cloudflare-env.secrets.d.ts b/cloudflare-env.secrets.d.ts new file mode 100644 index 0000000..77af03e --- /dev/null +++ b/cloudflare-env.secrets.d.ts @@ -0,0 +1,44 @@ +// Worker secrets, declared by hand. +// +// `cloudflare-env.d.ts` is regenerated by `pnpm cf-typegen`, which reads +// wrangler.jsonc — and secrets are deliberately not in that file. Declaring +// them there would be silently wiped on the next regen, so they live here and +// merge into the same interface. +// +// Set per environment with: +// pnpm exec wrangler secret put --env staging +// pnpm exec wrangler secret put --env production +// For local `wrangler dev` / `next dev`, put them in `.dev.vars`. +// +// Optional on purpose: the app must typecheck and build in CI, which has no +// secrets. Each read site is responsible for its own fallback or failure. + +declare namespace Cloudflare { + interface Env { + /** HMAC salt for hashing visitor IPs before they reach D1. */ + TOOLS_IP_SALT?: string + /** OpenRouter API key for the chatbot tool. Never sent to the client. */ + OPENROUTER_API_KEY?: string + /** Set to 'true' to allow the paid model tier as a last-resort fallback. */ + OPENROUTER_ALLOW_PAID?: string + /** + * Raises the per-visitor daily crawl limit. Local development only — + * localhost has no CF-Connecting-IP, so every request shares one counter + * and three crawls exhausts the day. Leave unset in staging and production. + */ + TOOLS_SESSIONS_PER_IP?: string + /** + * Lets an embed answer from any Origin. Local development only — a key is + * bound to the site it was minted for, so a widget cannot otherwise be + * previewed from localhost. Leave unset in staging and production: without + * the binding, a public key works from anywhere. + */ + TOOLS_EMBED_ALLOW_ANY_ORIGIN?: string + /** + * Transactional email provider key. Unset today, which makes + * lib/tools/email.ts log verification codes instead of sending them. + * Setting it also requires SPF and DKIM records on epyc.in. + */ + RESEND_API_KEY?: string + } +} diff --git a/components/sections/chatbot-tool.tsx b/components/sections/chatbot-tool.tsx new file mode 100644 index 0000000..f2b6872 --- /dev/null +++ b/components/sections/chatbot-tool.tsx @@ -0,0 +1,1109 @@ +'use client' + +import { useEffect, useRef, useState } from 'react' +import { cn } from '@/lib/cn' +import { Button } from '@/components/ui/button' +import { Container } from '@/components/ui/container' +import { Disc } from '@/components/ui/disc' +import { Field, Input } from '@/components/ui/form' +import { Pill } from '@/components/ui/pill' +import { Section } from '@/components/ui/section' +import { SectionHeading } from '@/components/ui/section-heading' +import { Plus, Sparkle } from '@/components/icons' +import { suggestedQuestions } from '@/data/buyer-questions' +import { readSSE } from '@/lib/tools/sse-client' + +/** + * The AI chatbot tool: paste a URL, we read the site, you chat with it. + * + * Wired to POST /api/tools/chatbot/crawl and /message (both stream Server-Sent + * Events) and GET /api/tools/chatbot/diagnosis. Flow and screen contents: + * docs/ai-chatbot-flow.md. + * + * The report's three measured scores are stored when the crawl finishes, so + * they are always ready. Answerability and Coverage are scored in the + * background on message one; if they have not landed yet the panel renders + * without them and polls briefly rather than blocking on a spinner. + */ + +type Phase = 'idle' | 'crawling' | 'chat' | 'report' | 'empty' | 'blocked' + +const MAX_MESSAGES = 8 + +type Msg = { from: 'bot' | 'you'; text: string; miss?: boolean; pending?: boolean } +type CrawledPage = { url: string; title: string } + +export function ChatbotTool() { + const [phase, setPhase] = useState('idle') + const [url, setUrl] = useState('') + const [error, setError] = useState(null) + + const [status, setStatus] = useState('') + const [pages, setPages] = useState([]) + const [total, setTotal] = useState(0) + + const [session, setSession] = useState<{ id: string; host: string; pages: number } | null>(null) + const [signals, setSignals] = useState | null>(null) + + const [messages, setMessages] = useState([]) + const [draft, setDraft] = useState('') + const [sending, setSending] = useState(false) + const [used, setUsed] = useState(0) + + const threadEnd = useRef(null) + + async function startCrawl(force = false) { + const target = url.trim() + if (!target) { + setError('Enter a website address.') + return + } + + setError(null) + setPages([]) + setTotal(0) + setStatus('') + setMessages([]) + setUsed(0) + setPhase('crawling') + + try { + const res = await fetch('/api/tools/chatbot/crawl', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ url: target, force }), + }) + + // Validation and cap failures come back as plain JSON, not a stream. + if (!res.ok && res.headers.get('content-type')?.includes('application/json')) { + const body = (await res.json()) as { error?: string } + setError(body.error ?? 'Something went wrong.') + setPhase('idle') + return + } + + await readSSE(res, (event, data) => { + if (event === 'status') setStatus(String(data.message ?? '')) + if (event === 'page') { + setPages((p) => [...p, { url: String(data.url), title: String(data.title ?? '') }]) + setTotal(Number(data.total ?? 0)) + } + if (event === 'error') { + setError(String(data.message ?? 'We could not read that site.')) + setPhase('idle') + } + if (event === 'done') { + const host = hostOf(target) + setSession({ id: String(data.sessionId), host, pages: Number(data.pages ?? 0) }) + setSignals((data.signals as Record) ?? null) + + const st = String(data.status) + if (st === 'ready') { + setPhase('chat') + setMessages([ + { + from: 'bot', + text: `I've read ${data.readablePages ?? data.pages} pages of ${host}. Ask me anything a customer might ask.`, + }, + ]) + } else if (st === 'empty') { + setPhase('empty') + } else { + setPhase('blocked') + } + } + }) + } catch { + setError('We could not reach that site. Try another address.') + setPhase('idle') + } + } + + async function send(text: string) { + if (!session || sending || !text.trim()) return + + setDraft('') + setSending(true) + setMessages((m) => [...m, { from: 'you', text }, { from: 'bot', text: '', pending: true }]) + scrollSoon() + + try { + const res = await fetch('/api/tools/chatbot/message', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ sessionId: session.id, message: text }), + }) + + if (!res.ok && res.headers.get('content-type')?.includes('application/json')) { + const body = (await res.json()) as { error?: string; capped?: boolean } + setMessages((m) => replaceLast(m, { from: 'bot', text: body.error ?? 'Something went wrong.' })) + if (body.capped) setPhase('report') + return + } + + let answer = '' + await readSSE(res, (event, data) => { + if (event === 'delta') { + answer += String(data.text ?? '') + setMessages((m) => replaceLast(m, { from: 'bot', text: answer, miss: looksLikeMiss(answer) })) + scrollSoon() + } + if (event === 'error') { + answer += `\n\n${String(data.message ?? '')}` + setMessages((m) => replaceLast(m, { from: 'bot', text: answer })) + } + if (event === 'done') { + setUsed(Number(data.messagesUsed ?? 0)) + if (data.capped) setPhase('report') + } + }) + } catch { + setMessages((m) => + replaceLast(m, { from: 'bot', text: 'I lost that answer. Try asking again.' }), + ) + } finally { + setSending(false) + scrollSoon() + } + } + + function scrollSoon() { + requestAnimationFrame(() => threadEnd.current?.scrollIntoView({ behavior: 'smooth', block: 'end' })) + } + + return ( + <> + {phase === 'idle' && ( + void startCrawl()} + /> + )} + + {phase === 'crawling' && ( + + )} + + {phase === 'chat' && session && ( + setPhase('report')} + /> + )} + + {phase === 'report' && session && ( + void startCrawl(true)} + /> + )} + + {phase === 'empty' && ( + void startCrawl(true)} + /> + )} + + {phase === 'blocked' && ( + setPhase('idle')} /> + )} + + ) +} + +/* ------------------------------------------------------------------ helpers */ + +function hostOf(input: string): string { + try { + return new URL(/^https?:\/\//i.test(input) ? input : `https://${input}`).host + } catch { + return input + } +} + +function replaceLast(messages: Msg[], next: Msg): Msg[] { + return [...messages.slice(0, -1), next] +} + +/** Highlight an honest miss — the answer the whole report is built from. */ +function looksLikeMiss(text: string): boolean { + return /couldn'?t find|could not find|isn'?t mentioned|is not mentioned|no mention|doesn'?t say|does not say/i.test( + text, + ) +} + +/* --------------------------------------------------------------- 1. Idle */ + +function IdleScreen({ + url, + error, + onChange, + onSubmit, +}: { + url: string + error: string | null + onChange: (v: string) => void + onSubmit: () => void +}) { + return ( +
+ +
+ Free · No signup · About 30 seconds + +

+ Can an AI actually read +
+ your website? +

+ +

+ Paste your address. We read up to 20 pages, build a chatbot from what we find, and + show you the 10 questions a buyer asks that your site cannot answer. +

+ +
{ + e.preventDefault() + onSubmit() + }} + > + + onChange(e.target.value)} + autoComplete="url" + /> + + {/* `Button` spreads its rest props after its own `type`, so this + overrides the default `type="button"` — no nested button. */} + +
+ + {error ? ( +

+ {error} +

+ ) : ( +

+ We only read pages your robots.txt allows. Nothing is published anywhere. +

+ )} +
+ +
+ {[ + { n: '01', t: 'We read it', b: 'Up to 20 pages, the way an AI assistant would.' }, + { n: '02', t: 'You question it', b: 'Eight questions to a bot that only knows your site.' }, + { n: '03', t: 'You get the report', b: 'Five scores, every one backed by your own pages.' }, + ].map((s) => ( +
+ {s.n} +

{s.t}

+

{s.b}

+
+ ))} +
+
+
+ ) +} + +/* ----------------------------------------------------------- 2. Crawling */ + +function CrawlingScreen({ + host, + status, + pages, + total, +}: { + host: string + status: string + pages: CrawledPage[] + total: number +}) { + const pct = total ? Math.round((pages.length / total) * 100) : 8 + + return ( +
+ +
+ + Reading your site + + +
+
+ + {total ? `${pages.length} of ${total} pages` : 'Getting started'} + +
+
+
+
+
+ +
    + {pages.map((p, i) => ( +
  • + + {p.url} +
  • + ))} +
+ + {status &&

{status}

} +
+ +
+ ) +} + +/* --------------------------------------------------------------- 3. Chat */ + +function ChatScreen({ + host, + messages, + draft, + sending, + left, + threadEnd, + onDraft, + onSend, + onSkip, +}: { + host: string + messages: Msg[] + draft: string + sending: boolean + left: number + threadEnd: React.RefObject + onDraft: (v: string) => void + onSend: (text: string) => void + onSkip: () => void +}) { + return ( +
+ +
+
+
+ Chatting with + {host} +
+ + {left} of {MAX_MESSAGES} questions left + +
+ +
+ {messages.map((m, i) => ( + + ))} +
+
+ +
+ Try asking +
+ {suggestedQuestions.map((q) => ( + + ))} +
+
+ +
{ + e.preventDefault() + onSend(draft) + }} + > + + onDraft(e.target.value)} + /> + + +
+ + +
+ +
+ ) +} + +function Bubble({ msg }: { msg: Msg }) { + const isYou = msg.from === 'you' + return ( +
+
+ {msg.miss && ( + Not on the site + )} + {msg.pending && !msg.text ? Reading your site… : msg.text} +
+
+ ) +} + +/* ------------------------------------------------------------- 4. Report */ + +type Verdict = 'pass' | 'weak' | 'fail' + +type Diagnosis = { + answerability: { answered: number; total: number; unanswered: { id: string; question: string }[] } | null + coverage: { present: string[]; missing: string[] } | null + structure: { verdict: Verdict; headline: string; evidence: string[] } + crawlability: { + verdict: Verdict + headline: string + checks: { label: string; pass: boolean; detail: string }[] + } + specificity: { verdict: Verdict; headline: string; examples: { quote: string; url: string }[] } + partial: boolean +} + +function ReportScreen({ + sessionId, + host, + onRecrawl, +}: { + sessionId: string + host: string + onRecrawl: () => void +}) { + const [diagnosis, setDiagnosis] = useState(null) + const [failed, setFailed] = useState(false) + + useEffect(() => { + let cancelled = false + let attempts = 0 + + // The judged half is scored in the background on message one. It is + // usually done well before the visitor gets here; if not, poll briefly + // rather than making them look at a spinner or a half-report. + async function load() { + try { + const res = await fetch(`/api/tools/chatbot/diagnosis?sessionId=${sessionId}`) + const body = (await res.json()) as { ok: boolean; diagnosis?: Diagnosis } + if (cancelled) return + + const d = body.diagnosis + // A diagnosis missing its measured half would take the whole page + // down when rendered. The three measured scores are written when the + // crawl finishes, so their absence means something is wrong upstream — + // show the failure state rather than a half-report that crashes. + const complete = Boolean(d?.structure && d.crawlability && d.specificity) + + if (body.ok && d && complete) { + setDiagnosis(d) + // Scoring takes ~20s against a full corpus, so poll for ~40s before + // giving up and leaving the three measured scores on screen. + if (d.partial && attempts < 20) { + attempts++ + setTimeout(load, 2000) + } + } else { + setFailed(true) + } + } catch { + if (!cancelled) setFailed(true) + } + } + + void load() + return () => { + cancelled = true + } + }, [sessionId]) + + if (failed) { + return ( +
+ +
+

We couldn’t put your report together

+

+ Reading {host} worked, but scoring it did not. Running it again usually fixes this. +

+
+ + +
+
+
+
+ ) + } + + const a = diagnosis?.answerability + + return ( + <> +
+ +
+ {host} + + {a ? ( + <> +

+ + {a.answered} + {' '} + of {a.total} +

+

buyer questions your website can answer

+

+ {a.unanswered.length === 0 + ? 'Your site answers everything a buyer asks before getting in touch. That is rare.' + : 'The bot was limited by what your site says, not by the bot. Here is what it could not find.'} +

+ + {a.unanswered.length > 0 && ( +
    + {a.unanswered.map((q) => ( +
  • + + {q.question} +
  • + ))} +
+ )} + + ) : ( + <> +

…

+

Scoring your site

+

+ Checking your pages against the ten questions a buyer asks. The rest of your + report is below. +

+ + )} +
+
+
+ + {diagnosis && ( +
+ +
+ + The rest of the report + +
+ {diagnosis.coverage && ( + `Missing: ${m}`) + : ['Every page type a buyer looks for is present'] + } + /> + )} + + `${c.pass ? '✓' : '✗'} ${c.label} — ${c.detail}`, + )} + /> + `“${e.quote}” — ${pathOnly(e.url)}`, + )} + /> +
+
+
+
+ )} + +
+ +
+

+ {a && a.answered >= 8 + ? 'You’re most of the way there.' + : 'This is a content problem, not a bot problem.'} +

+

+ Every question above is one a buyer asks before they get in touch. We rebuild sites + so the answers are on the page. +

+
+ + +
+
+
+
+ + + + ) +} + +/** + * Keep the bot: capture an email, mint a key, hand over the snippet. + * + * The email is not verified yet — nothing can send mail from this app. When + * that exists, a code step slots in before the snippet appears and nothing + * else here changes. + */ +function ClaimEmbed({ sessionId, host }: { sessionId: string; host: string }) { + const [step, setStep] = useState<'email' | 'code' | 'done'>('email') + const [email, setEmail] = useState('') + const [code, setCode] = useState('') + const [snippet, setSnippet] = useState(null) + const [error, setError] = useState(null) + const [notice, setNotice] = useState(null) + const [busy, setBusy] = useState(false) + const [copied, setCopied] = useState(false) + + /** Step one: ask for a code. */ + async function sendCode() { + if (busy || !email.trim()) return + setBusy(true) + setError(null) + + try { + const res = await fetch('/api/tools/chatbot/verify', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ action: 'send', sessionId, email: email.trim() }), + }) + const body = (await res.json()) as { ok: boolean; error?: string; stubbed?: boolean } + + if (body.ok) { + setStep('code') + setNotice( + body.stubbed + ? 'Email sending is not configured yet — the code is printed in the dev server console.' + : `We sent a 6-digit code to ${email.trim()}. It expires in 10 minutes.`, + ) + } else { + setError(body.error ?? 'Something went wrong. Try again.') + } + } catch { + setError('Something went wrong. Try again.') + } finally { + setBusy(false) + } + } + + /** Step two: check the code, then mint the key. */ + async function verifyAndClaim() { + if (busy || code.trim().length !== 6) return + setBusy(true) + setError(null) + + try { + const checked = await fetch('/api/tools/chatbot/verify', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ action: 'check', sessionId, email: email.trim(), code: code.trim() }), + }) + const checkBody = (await checked.json()) as { ok: boolean; error?: string } + if (!checkBody.ok) { + setError(checkBody.error ?? 'That code is not right.') + return + } + + const res = await fetch('/api/tools/chatbot/embed', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ sessionId, email: email.trim() }), + }) + const body = (await res.json()) as { ok: boolean; snippet?: string; error?: string } + if (body.ok && body.snippet) { + setSnippet(body.snippet) + setStep('done') + setNotice(null) + } else { + setError(body.error ?? 'Something went wrong. Try again.') + } + } catch { + setError('Something went wrong. Try again.') + } finally { + setBusy(false) + } + } + + async function copy() { + if (!snippet) return + try { + await navigator.clipboard.writeText(snippet) + setCopied(true) + setTimeout(() => setCopied(false), 2000) + } catch { + setError('Copy failed — select the code and copy it manually.') + } + } + + return ( +
+ +
+ + Keep the bot + + + {step === 'done' && snippet ? ( + <> +

+ Paste this into your site’s HTML, just before the closing{' '} + </body> tag. The bot appears in + the corner of every page it’s on. +

+ +
+                {snippet}
+              
+ +
+ + + Works on {host} only. Free, and it answers from the pages we just read. + +
+ + ) : step === 'code' ? ( + <> +

+ Enter the 6-digit code we sent to {email}. +

+ +
{ + e.preventDefault() + void verifyAndClaim() + }} + > + + setCode(e.target.value.replace(/\D/g, ''))} + /> + + +
+ + + + ) : ( + <> +

+ Put this same bot on {host}. It answers your visitors from your own pages, free. + We’ll email you a code to confirm the address. +

+ +
{ + e.preventDefault() + void sendCode() + }} + > + + setEmail(e.target.value)} + autoComplete="email" + /> + + +
+ + )} + + {notice &&

{notice}

} + + {error && ( +

+ {error} +

+ )} +
+
+
+ ) +} + +function ScoreRow({ + name, + verdict, + headline, + lines, +}: { + name: string + verdict: Verdict + headline: string + lines: string[] +}) { + const tone = { fail: 'text-crimson', weak: 'text-ink', pass: 'text-teal-deep' }[verdict] + + return ( +
+
+

{name}

+ {headline} +
+ {lines.length > 0 && ( +
    + {lines.map((line, i) => ( +
  • +
  • + ))} +
+ )} +
+ ) +} + +function pathOnly(url: string): string { + try { + return new URL(url).pathname + } catch { + return url + } +} + +/* -------------------------------------------------------------- 5. Empty */ + +function EmptyScreen({ + host, + signals, + pages, + onRetry, +}: { + host: string + signals: Record | null + pages: number + onRetry: () => void +}) { + return ( +
+ +
+ + We could not read your site + + +

+ We reached your pages, but they returned almost no readable text — usually because + everything is drawn by JavaScript after the page loads. An AI assistant reading your + site sees what we saw: an empty page. +

+ +
+ What we found + {[ + ['Sitemap', signals?.sitemapFound ? 'Found' : 'Not found'], + ['robots.txt', signals?.robotsBlockedAll ? 'Blocks crawling' : 'Allows crawling'], + ['Pages we could open', String(pages)], + ['Readable text without JavaScript', 'Almost none'], + ].map(([k, v]) => ( +
+ {k} + {v} +
+ ))} +
+ +
+ + +
+
+
+
+ ) +} + +/* ------------------------------------------------------------ 6. Blocked */ + +function BlockedScreen({ + host, + signals, + onBack, +}: { + host: string + signals: Record | null + onBack: () => void +}) { + return ( +
+ +
+ + We couldn’t reach that site + + +

+ {signals?.robotsBlockedAll + ? 'Your robots.txt blocks automated readers from every page. Search engines and AI assistants hit the same wall we just did.' + : 'Nothing came back that we could read. The site may be down, very slow, or blocking automated readers.'} +

+ +
+ + +
+
+
+
+ ) +} diff --git a/data/buyer-questions.ts b/data/buyer-questions.ts new file mode 100644 index 0000000..a89b3a3 --- /dev/null +++ b/data/buyer-questions.ts @@ -0,0 +1,88 @@ +/** + * The 10 questions a real buyer asks before they get in touch. + * + * These drive the Answerability score — the headline number on the report — + * and 4 of them render as clickable prompts in the chat composer, so the final + * score reads as a recap of failures the visitor already watched rather than a + * claim they have to accept. + * + * Deliberately a flat, editable list: change the wording here and both the + * score and the prompt chips follow. Keep them buyer-voiced ("what does it + * cost"), never company-voiced ("pricing page exists"), and keep them + * stage-agnostic — they are asked of a two-person studio and a listed company + * alike. + * + * DRAFT — needs sign-off before the report ships (see docs/ai-chatbot-plan.md, + * "Decisions needed" #2). + */ + +export type BuyerQuestion = { + /** Stable key. Never renumber — the score history is keyed on it. */ + id: string + /** Asked of the bot verbatim, in the buyer's voice. */ + question: string + /** Shown on the report next to a failed question, to explain the miss. */ + looksLike: string + /** The 4 marked true become the composer's suggested prompts. */ + suggested?: boolean +} + +export const buyerQuestions: readonly BuyerQuestion[] = [ + { + id: 'what-you-do', + question: 'What exactly does this company do?', + looksLike: 'A plain description of the service, not a slogan.', + suggested: true, + }, + { + id: 'who-for', + question: 'Who is it for — what kind of company or customer?', + looksLike: 'A named audience, industry, or company size.', + suggested: true, + }, + { + id: 'what-it-costs', + question: 'What does it cost, or how is pricing decided?', + looksLike: 'A number, a range, a starting price, or an explained model.', + suggested: true, + }, + { + id: 'how-long', + question: 'How long does a typical project take?', + looksLike: 'A timeline in weeks or months, even an approximate one.', + }, + { + id: 'the-process', + question: 'What happens between first contact and finished work?', + looksLike: 'Named stages, or a described way of working.', + }, + { + id: 'who-worked-with', + question: 'Who have they worked with before?', + looksLike: 'Named clients, logos with context, or case studies.', + suggested: true, + }, + { + id: 'what-results', + question: 'What results have they actually produced?', + looksLike: 'Outcomes with numbers attached, not adjectives.', + }, + { + id: 'why-them', + question: 'What makes them different from the alternatives?', + looksLike: 'A specific claim a competitor could not copy verbatim.', + }, + { + id: 'who-does-work', + question: 'Who actually does the work?', + looksLike: 'Team size, names, or experience — proof of a real team.', + }, + { + id: 'how-to-start', + question: 'How do I get in touch, and what happens next?', + looksLike: 'A contact route and a stated next step.', + }, +] as const + +/** The 4 that seed the composer. Kept as a derivation so the list stays single-source. */ +export const suggestedQuestions = buyerQuestions.filter((q) => q.suggested) diff --git a/db/migrations/0003_tool_sessions.sql b/db/migrations/0003_tool_sessions.sql new file mode 100644 index 0000000..9fae0bf --- /dev/null +++ b/db/migrations/0003_tool_sessions.sql @@ -0,0 +1,60 @@ +-- Free-tool sessions: the AI chatbot demo at /tools/ai-chatbot, and later the +-- Website Grader and llms.txt Generator (hence `tool` rather than a per-tool +-- table). Written by app/api/tools/chatbot/*, D1 binding `DB`. +-- +-- SQLite, not Postgres — D1 is SQLite. Idempotent, so it is safe to re-run. +-- Shares the `DB` binding, and therefore the database, with contact_submissions. +-- +-- `tool_embeds` is deliberately absent: the embeddable widget is phase two and +-- ships with its own migration. See docs/ai-chatbot-plan.md. +-- +-- Apply with --remote, or you only touch the local simulated database: +-- pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --remote \ +-- --file=db/migrations/0003_tool_sessions.sql + +-- One row per demo session. +CREATE TABLE IF NOT EXISTS tool_sessions ( + id TEXT PRIMARY KEY, -- crypto.randomUUID(), client-visible + tool TEXT NOT NULL CHECK (tool IN ('chatbot', 'grader', 'llms-txt')), + target_url TEXT NOT NULL, + host TEXT NOT NULL, -- normalised hostname; drives 24h crawl reuse + ip_hash TEXT NOT NULL, -- HMAC-SHA256(ip, TOOLS_IP_SALT). Never a bare hash. + status TEXT NOT NULL CHECK (status IN ('crawling', 'ready', 'empty', 'failed')), + pages_crawled INTEGER NOT NULL DEFAULT 0, + messages_used INTEGER NOT NULL DEFAULT 0, + diagnosis_json TEXT, -- scored once, in the background + transcript_json TEXT, -- what visitors actually ask + email TEXT, -- set only on interest capture + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +-- Extracted corpus, one row per page. +CREATE TABLE IF NOT EXISTS tool_pages ( + session_id TEXT NOT NULL REFERENCES tool_sessions(id), + url TEXT NOT NULL, + title TEXT, + text TEXT NOT NULL, + meta_json TEXT, -- heading depths, word count, js-empty flag + PRIMARY KEY (session_id, url) +); + +-- Atomic daily counters. key: 'global-messages' | 'ip:' | later 'embed:' +CREATE TABLE IF NOT EXISTS tool_counters ( + day TEXT NOT NULL, -- UTC YYYY-MM-DD + key TEXT NOT NULL, + n INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (day, key) +); + +-- Email captures. kind: 'embed' (phase-two gate) | 'model:' +CREATE TABLE IF NOT EXISTS tool_interest ( + id TEXT PRIMARY KEY, + session_id TEXT REFERENCES tool_sessions(id), + kind TEXT NOT NULL, + email TEXT, + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +-- Crawl reuse looks up the most recent ready session for a host. +CREATE INDEX IF NOT EXISTS idx_tool_sessions_host_created + ON tool_sessions (host, created_at DESC); diff --git a/db/migrations/0004_tool_embeds.sql b/db/migrations/0004_tool_embeds.sql new file mode 100644 index 0000000..4ae2d36 --- /dev/null +++ b/db/migrations/0004_tool_embeds.sql @@ -0,0 +1,34 @@ +-- Claimed embeds: one row per widget running on a third-party site. +-- +-- Written by app/api/tools/chatbot/embed/route.ts, read on every embed +-- message. Separate from 0003 because the widget is a later decision — see +-- docs/ai-chatbot-plan.md. +-- +-- The key is PUBLIC by design. It sits in the HTML of the host site, so it +-- cannot be a secret. Security comes from binding, not concealment: a key +-- minted for acme.com only answers requests whose Origin is acme.com, and the +-- per-key daily cap bounds the damage if someone forges the Origin header from +-- a non-browser client. +-- +-- Apply with --remote, or you only touch the local simulated database: +-- pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --remote \ +-- --file=db/migrations/0004_tool_embeds.sql + +CREATE TABLE IF NOT EXISTS tool_embeds ( + key TEXT PRIMARY KEY, -- 'ek_live_…', public, lives in host HTML + session_id TEXT NOT NULL REFERENCES tool_sessions(id), + bound_host TEXT NOT NULL, -- apex host; www is accepted implicitly + email TEXT, -- who claimed it + status TEXT NOT NULL DEFAULT 'active' CHECK (status IN ('active', 'revoked')), + attribution INTEGER NOT NULL DEFAULT 1,-- "Powered by EPYC", not removable on free + crawled_at TEXT NOT NULL, -- shown in the widget, drives recrawl prompts + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP, + last_message_at TEXT +); + +-- One embed per session, and a fast lookup when someone re-claims the same host. +CREATE UNIQUE INDEX IF NOT EXISTS idx_tool_embeds_session ON tool_embeds (session_id); +CREATE INDEX IF NOT EXISTS idx_tool_embeds_host ON tool_embeds (bound_host); + +-- A claimed embed's corpus must survive: the live bot answers from it. Any +-- pruning job must skip pages whose session has a row here. diff --git a/db/migrations/0005_tool_verifications.sql b/db/migrations/0005_tool_verifications.sql new file mode 100644 index 0000000..ffbfead --- /dev/null +++ b/db/migrations/0005_tool_verifications.sql @@ -0,0 +1,29 @@ +-- Email verification codes for claiming an embed. +-- +-- A pending code, never the code itself: `code_hash` is an HMAC, so a database +-- read cannot hand anyone a working code. Six digits is only a million +-- possibilities, which is why the attempt limit below is load-bearing rather +-- than decorative. +-- +-- Daily caps (per email, per session) are enforced through tool_counters +-- rather than by counting rows here — that counter is a single atomic +-- statement and already exists. See lib/tools/chatbot/verification.ts. +-- +-- Apply with --remote, or you only touch the local simulated database: +-- pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --remote \ +-- --file=db/migrations/0005_tool_verifications.sql + +CREATE TABLE IF NOT EXISTS tool_verifications ( + id TEXT PRIMARY KEY, + session_id TEXT NOT NULL REFERENCES tool_sessions(id), + email TEXT NOT NULL, + code_hash TEXT NOT NULL, -- HMAC-SHA256(code, TOOLS_IP_SALT) + expires_at TEXT NOT NULL, -- ISO. 10 minutes from issue + attempts INTEGER NOT NULL DEFAULT 0, -- invalidated past 5 + consumed_at TEXT, -- set once; a code is single use + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +-- Checking a code looks up the newest live one for this session and email. +CREATE INDEX IF NOT EXISTS idx_tool_verifications_lookup + ON tool_verifications (session_id, email, created_at DESC); diff --git a/docs/ai-chatbot-architecture.md b/docs/ai-chatbot-architecture.md new file mode 100644 index 0000000..1f20fd6 --- /dev/null +++ b/docs/ai-chatbot-architecture.md @@ -0,0 +1,179 @@ +# AI Chatbot — Architecture + +**Companion docs:** [`ai-chatbot-plan.md`](./ai-chatbot-plan.md) (scope, order of work, approvals) · [`ai-chatbot-tech.md`](./ai-chatbot-tech.md) (route contracts, SQL, prompts, build reference) + +This document explains **why the system is shaped the way it is** — what runs where, what was rejected and on what grounds, and how phase one connects to the phase-two target without rework. + +--- + +## 1. What the thing is + +A lead-gen diagnostic that happens to use a chatbot. A visitor pastes their website address; we read up to 20 pages of it, build a chatbot from that text, let them ask 8 questions, then show a scored report on why the bot struggled — framed as **their content problem**, evidenced from their own pages. + +The conversion is the **website rebuild project**, not a chatbot build. Every surface routes to the rebuild conversation. The bot is best-effort: no sandbagging, no scolding panel alongside the chat. The critique lives in the report, after the conversation, never in the bot's voice. + +--- + +## 2. Phase-one system shape + +Everything runs inside the existing Next.js app on Cloudflare Workers. One new dependency family (AI SDK + OpenRouter provider), one new migration, no new services. + +``` +Browser /tools/ai-chatbot + │ + │ 1. POST /api/tools/chatbot/crawl (SSE — live progress) + ▼ +Next.js Worker ──── fetch robots.txt ─────────► prospect site + │ ──── fetch sitemap(s) ────────► + │ ──── fetch ≤20 pages ─────────► + │ extract text, truncate + ├──► D1 tool_sessions / tool_pages / tool_counters + │ + │ 2. POST /api/tools/chatbot/message (SSE — streamed tokens) + ▼ +Next.js Worker ──► corpus from D1 + history ──► OpenRouter (Lightning free) + │ └─ paid fallback on 429 + │ after message 1: waitUntil(score) ────────► OpenRouter (one JSON call) + ├──► D1 tool_sessions.diagnosis_json + │ + │ 3. GET /api/tools/chatbot/diagnosis (reads the stored JSON) + │ 4. POST /api/tools/chatbot/interest (email capture — phase-two gate) + ▼ +Report panel + two CTAs +``` + +**Three architectural calls carry this design.** + +### 2.1 The corpus goes in the context window. No retrieval layer. + +At 20 pages × ~4k tokens the whole site is roughly an 80k-token block, which fits one request. A vector store buys nothing at that size: it adds embeddings, an index, a query-time embed call, and a chunking strategy, in exchange for savings measured in cents on a single session. + +**Confirmed, not assumed** (OpenRouter live model list, 14 Aug 2026): `nvidia/nemotron-3.5-lightning:free` carries a **1,000,000-token context window** at $0 in / $0 out. This was the load-bearing assumption of the whole design and it holds with 12× headroom. + +The fallback chain is free at every tier — Lightning `:free` → Super `:free` (1M window) → Ultra `:free` (512K), with a paid model configured but disabled behind a flag. Phase one therefore carries **no per-message cost**, which is why the caps in §4 are described as abuse control rather than budget control. The open risk is not price but quota shape: if OpenRouter meters free usage per *account* rather than per model, the cascade is decorative and the paid flag becomes tier 4. Test with concurrent sessions before launch. + +It also *costs* quality. Chunked retrieval hands the model fragments; "what do you do" and "what does it cost" answer better from whole pages. Retrieval here would be a cost optimisation sold as a quality one. + +Retrieval becomes correct at tenant volume — see §5. + +### 2.2 The crawl runs inline in the route handler, streamed as SSE. No queue, no second worker. + +The source spec routed the crawl through `CRAWL_QUEUE` → `workers/tool-crawler` → poll for status. That is the same asynchronous shape the spec itself rejected Cloudflare AI Search for ("indexing runs as an async sync job; the demo needs an answer in seconds"). + +Doing it inline deletes: one worker, one queue declared six times across two configs, a `queued|crawling|ready|failed` state machine, and a polling endpoint that never made it into the route list. It also buys honest progress copy — "reading /pricing" instead of a spinner. + +It costs nothing in headroom: ~22 subrequests, almost all I/O wait; the CPU work is HTML→text over ≤500KB × 20 pages, comfortably inside the Workers budget. A hard 20-second wall-clock deadline bounds the worst case — 12 pages of content beats one page that never loads. + +Two dependencies worth naming rather than assuming. The **subrequest ceiling is 1000 on paid Workers plans and 50 on Free** — this repo already uses Queues and D1, which are paid-plan features, so 1000 applies; confirm before relying on it. And **SSE must survive the OpenNext Cloudflare adapter**. Streaming responses are supported, but this is the one assumption that would change the UX if wrong, so it gets a 30-minute throwaway-route spike before step 3 depends on it. If responses turn out to be buffered, the fallback is a determinate progress bar — the architecture is unchanged, only the "reading /pricing" polish is lost. + +**Security falls out of this for free.** `global_fetch_strictly_public` is set on the main worker's `wrangler.jsonc` and blocks fetches to private and internal addresses. Doing the crawl in a separate worker would have meant setting that flag there too — the sibling precedent (`workers/contact-webhook/wrangler.jsonc`) ships `nodejs_compat` only. Doing it inline means the primary SSRF mitigation is already where the fetching happens. + +### 2.3 The report is scored quietly after message one. + +Three of the five dimensions — Structure, Crawlability, Specificity — are deterministic from crawl metadata and cost nothing. Two — Answerability, Coverage — need judgement, and they run as **one** structured JSON call, not eleven. + +That call fires in the background via `waitUntil` on the first message and stores its result on the session. By the time the visitor hits the cap at message 8, the panel renders instantly instead of stalling ten seconds at the exact moment we ask for the click. Gating on first message rather than on crawl completion means visitors who bounce cost nothing. + +If the model call fails, the three deterministic scores still render. A partial report converts; an error state does not. + +**The empty-corpus path needs its own trigger.** A site that yields almost no readable text skips the chat entirely and goes straight to the report — which means message one never happens, and a report scored on message one would never be scored at all. For that path, scoring runs at the end of the crawl instead. Answerability is 0 of 10 by construction (there is no text to answer from) so no model call is needed, and the report leads with the crawlability blocker, which is the finding that matters anyway. + +**One judgement call left open on the model.** Lightning is a small MoE — 3B active of 30B — which is the right shape for "answer from the text in front of you". The scoring call is a different job: judgement across the whole corpus, producing the headline number shown to a prospect, once per session rather than eight times. That is the one call where paying is cheap and being wrong is expensive. Build on Lightning, compare against a larger model across ten real sites before launch, and spend the cents if the gap is visible. + +--- + +## 3. Rejected alternatives, with grounds + +| Rejected | Why | +|---|---| +| **Cloudflare AI Search (AutoRAG) with its own crawler** | Website data source is documented as "a domain you own" — arbitrary prospect URLs fall outside intended use. Indexing is async. Workers Free caps at 500 pages/day and 100 instances/account; a public demo would want one instance per visitor URL. `blocked_by_robots_txt` and `subdomains_not_allowed` would fire constantly against third-party sites. *(AI Search is the right tool for a chatbot on epyc.in itself — own domain, one instance, sitemap already published. Separate project.)* | +| **AutoRAG fed from R2 instead of its own crawler** | Kills two of the four objections above, but not the async one. Under R2 it is still a job, not a function call: upload markdown, then wait for AutoRAG to notice, chunk, embed, index. The demo's promise is "paste a URL, talk to your bot". | +| **Crawl4AI (Python + Playwright + Chromium)** | Strongest crawler of the options, and the right tool for a crawling product. Cost is architectural: cannot run on Workers, so a second runtime, a second deploy pipeline, a second thing to monitor, and an always-on container (512MB–1GB idling) to serve bursty demo traffic. Critically, `global_fetch_strictly_public` does not reach into a container, and a headless browser is a materially larger SSRF surface than `fetch` — it loads subresources, iframes and redirects. **Phase one takes arbitrary URLs from anonymous visitors, which is the worst possible threat model for that.** In phase two the crawl target is a domain someone claimed. The security argument for plain `fetch` is strongest exactly where we are now. | +| **Durable Object for rate limiting** | One atomic D1 statement gives the same guarantee (reject before any paid call, no race across concurrent Workers). We need three counter scopes — `global-messages`, `ip:`, later `embed:` — and a DO-per-chatbot covers only the third; D1 would be needed anyway. One mechanism, not two. Revisit at ~100× the planned volume. | +| **The Vercel Next.js chatbot template** | A full multi-user product: assumes Vercel OIDC and Blob, ships Auth.js and Neon Postgres for logged-in history when this tool is anonymous and D1 is already bound, routes models through AI Gateway when the model choice is OpenRouter-specific, and carries artifacts, document editing and file upload that are all unused. It has no crawl layer, which is the actual hard part. The one reusable piece is the AI SDK, installed on its own. | +| **`@assistant-ui/react` as specced ("composes into the existing shadcn/ui")** | This repo has no shadcn/ui. It has its own system: cva + Tailwind v4 `@theme` tokens documented in `DESIGN.md`, with `cn()` extended so custom `text-h2`-style utilities do not collide. Dropping assistant-ui in unmodified imports a second design vocabulary plus Radix plus shadcn CSS variables that do not exist here. **The restyle onto EPYC tokens is real work and is budgeted as such — it is why step 6 is the largest step.** If that step needs cutting, hand-building the thread is the lever: one thread, no history, no tool calls, no uploads. | +| **JS-rendered sites: render them properly** | Detect and score the failure instead. A site an AI cannot read *is the finding* — the most on-message thing the tool can say. Cloudflare Browser Rendering is tier two if drop-off data shows SPAs are a large share of submissions; it stays on Workers and is pay-per-use. Crawl4AI only if bot-blocking, rather than JS rendering, turns out to be the real blocker. | + +--- + +## 4. Data model, and why four tables + +Full DDL in [`ai-chatbot-tech.md`](./ai-chatbot-tech.md#d1-schema). + +| Table | Holds | Note | +|---|---|---| +| `tool_sessions` | one row per demo session, incl. `diagnosis_json`, `transcript_json`, `email` | `diagnosis_json` on the session is what makes "score once, render instantly" possible | +| `tool_pages` | extracted corpus, one row per page | pruned on a schedule; unclaimed sessions are demo exhaust | +| `tool_counters` | atomic daily counters | three scopes today: `global-messages`, `ip:`, and per-session message count | +| `tool_interest` | email captures | merges the spec's `tool_model_interest` with the phase-two embed waitlist: `kind` = `'embed'` \| `'model:'` | + +The spec's fifth table, `tool_embeds`, ships **with** the widget, not before it. + +Two corrections to the spec's own SQL, both load-bearing: + +- **The counter statement is broken as written.** `UPDATE ... WHERE n < ?` affects zero rows when no row exists for today, and the spec reads zero rows as "cap hit". Every counter would report exhausted on the first request after midnight. The `INSERT ... ON CONFLICT DO UPDATE ... WHERE n < ?` form keeps the single-statement atomicity and is correct on a cold day. +- **`ip_hash` needs a secret salt.** IPv4 is 2³² — a bare hash is a lookup table, not anonymisation. HMAC with a wrangler secret. This matters because the privacy line goes to a lawyer. + +--- + +## 5. The phase-two target, and why it is deferred + +The proposed phase-two architecture — self-hosted Crawl4AI → R2 → Cloudflare AI Search (AutoRAG) → Vectorize, generation on OpenRouter, nightly hash-diff re-crawl — is a good target. It is a **multi-tenant chatbot product's** architecture, and every component in it solves a phase-two problem: + +| Component | Problem it solves | Exists in phase one? | +|---|---|---| +| Vectorize / retrieval | per-message token cost at tenant volume | No — one session, 8 messages | +| AutoRAG | managed indexing of a growing corpus | No — and its async delay actively breaks the demo | +| R2 | durable corpus for a persistent widget | No — D1 holds it, pruned | +| Crawl4AI | SPA sites a paying customer needs working | No — that failure is the diagnostic's best finding | +| Nightly hash-diff re-crawl | staleness on someone's live site | No — the session lasts minutes | +| Durable Object | per-chatbot limits | No — there are no chatbots yet | + +Workers, D1 and OpenRouter are in both. That intersection **is** phase one's stack. + +### How the two connect + +The crawl runs once and feeds two consumers with different needs. Phase two attaches a limb; it does not replace one. + +``` +crawl output (clean text) + ├──► D1 ──► corpus in context ──► demo chat (instant) + │ └──► diagnosis scoring + └──► R2 ──► AutoRAG ──► Vectorize ──► embed chat (retrieval) + ▲ + └── only on embed claim +``` + +Three things fall out of that split for free: unclaimed sessions never touch AutoRAG (most visitors bounce, so we never pay to embed, store or index them, and the 100-instance ceiling stops being a scaling wall); the indexing delay lands between *claiming* an embed and the widget's first real visitor, where nobody is waiting; and the diagnosis stays on raw text beside the retrieval layer rather than behind it — four of the five dimensions cannot be computed from chunks at all. + +### Promoted into the phase-two target as written + +- **Content-hash diff on the nightly re-crawl, preserving IDs so embed codes never break.** Better than the spec's "manual recrawl is enough for launch", and cheap. +- **50 messages/day per embed** over the spec's 20, which the spec itself flagged as too low for a site with real traffic. + +### Carried as known risks for phase two + +- **Neuron budget is not fully escaped.** Generation moves to OpenRouter, but AutoRAG embeds the *query* on every `search()` using Workers AI. Neurons then scale with message volume, on the same shared account budget, failing all-tenants-at-once. +- **One AI Search instance, not one per chatbot** — a folder per tenant in one bucket, filtered at query time. That converts a hard 100-tenant ceiling into a filtering problem, and makes **tenant isolation the top security risk in the system**: a filter bug serves customer A's content from customer B's bot, on B's live site. That needs a test, not a code review. +- **Crawl4AI must be scale-to-zero and staggered.** Crawls are bursty — onboarding plus a nightly cron. An always-on Chromium container idling to serve that pattern is the spec's main cost objection and is avoidable. All tenants at once is a thundering herd on our own crawler and an impolite one on their sites. +- **Egress must be restricted at the network level** around any container — firewall or VPC rule blocking private ranges — plus private-IP blocking, DNS-rebind protection and redirect limits inside it. `global_fetch_strictly_public` does not apply there. +- **Third-party liability.** A bot on a client's live site answers their real customers. The system prompt's refusal to invent facts stops being a sales mechanic and becomes a liability question. Terms and a privacy line are launch blockers for the widget. + +The gate is the email count on the "want this on your site?" button: about a day of work, and it produces the same go/no-go signal that weeks of widget build would — the technique the spec already trusts for the paid-model question, applied one level up. + +--- + +## 6. What this reuses from the existing repo + +Nothing new at the platform level. + +| Existing | Reused for | +|---|---| +| D1 binding `DB` + `db/migrations/` | corpus storage, counters, captured emails | +| `app/api/contact/route.ts` | zod validation and route shape (validate → parse → write D1 → respond) | +| `global_fetch_strictly_public` in `wrangler.jsonc` | SSRF protection on the inline crawler | +| `components/ui/` + `DESIGN.md` tokens | the entire chat and report UI | +| OpenNext deploy pipeline, staging/production envs | deployment, unchanged | +| Manual `wrangler d1 execute --remote` convention | the new migration is a deploy-day checklist item, not automatic | + +The crawl engine written here is shared with the planned **Website Grader** and **llms.txt Generator**. Build it once with those two consumers in mind — a small, boring `lib/crawl/` module with a clean function boundary, not a route-local closure. diff --git a/docs/ai-chatbot-flow.md b/docs/ai-chatbot-flow.md new file mode 100644 index 0000000..a44c5b2 --- /dev/null +++ b/docs/ai-chatbot-flow.md @@ -0,0 +1,453 @@ +# AI Chatbot — Flow & Information Wireframe + +Black and white, no styling, no visual design. This shows **what information +appears on each screen** and **how the experience moves between them**, +including the paths where things go wrong. + +Visual design comes later — see `docs/ai-chatbot-tech.md` → Design. +Scope and build order: `docs/ai-chatbot-plan.md`. + +--- + +## 1. The whole flow + +``` + ┌─────────────────┐ + │ A · PASTE URL │ + └────────┬────────┘ + │ submit + ▼ + ┌─────────────────┐ + │ URL is checked │ + └────────┬────────┘ + ┌──────────────────┼──────────────────┐ + │ bad address │ ok │ over daily limit + ▼ ▼ ▼ + ┌───────────────┐ ┌─────────────────┐ ┌───────────────┐ + │ A1 · BAD URL │ │ B · READING │ │ A2 · COME │ + │ (inline error)│ │ THE SITE │ │ BACK │ + └───────┬───────┘ └────────┬────────┘ │ TOMORROW │ + │ │ └───────────────┘ + └──back to A │ + │ + ┌─────────────────────────┼─────────────────────────┐ + │ blocked / unreachable │ text found │ almost no text + ▼ ▼ ▼ +┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐ +│ B1 · CANNOT │ │ C · CHAT │ │ E · UNREADABLE │ +│ REACH IT │ │ (8 messages) │ │ SITE │ +└─────────────────┘ └────────┬────────┘ └────────┬────────┘ + │ 8 used, or │ + │ "show my report" │ skips chat + ▼ │ + ┌─────────────────┐ │ + │ D · REPORT │◀────────────────┘ + └────────┬────────┘ + │ + ┌─────────────┴─────────────┐ + ▼ ▼ + ┌───────────────┐ ┌─────────────────┐ + │ Talk about a │ │ Want this on │ + │ rebuild │ │ your site? │ + │ → /contact │ │ → F · EMAIL │ + └───────────────┘ └─────────────────┘ + + Re-crawl ("I fixed something, read it again") returns to B from D or E. +``` + +**The one-line version:** paste → we read → you chat → we score → two ways out. + +--- + +## 2. A · Paste a URL + +``` +┌──────────────────────────────────────────────────────────────┐ +│ [ EPYC logo ] Projects Blog Gallery … │ +│ │ +│ │ +│ Free · No signup · About 30 seconds │ +│ │ +│ Can an AI actually read your website? │ +│ │ +│ Paste your address. We read up to 20 pages, build a │ +│ chatbot from what we find, and show you the 10 questions │ +│ a buyer asks that your site cannot answer. │ +│ │ +│ ┌────────────────────────────────┐ ┌──────────────────┐ │ +│ │ yourcompany.com │ │ Read my site → │ │ +│ └────────────────────────────────┘ └──────────────────┘ │ +│ │ +│ We only read pages your robots.txt allows. │ +│ Nothing is published anywhere. │ +│ │ +│ ────────────────────────────────────────────────────────── │ +│ │ +│ (01) (02) (03) │ +│ We read it You question it You get │ +│ the report │ +│ Up to 20 pages, Eight questions to Five scores, │ +│ the way an AI a bot that only each backed │ +│ assistant would. knows your site. by your pages.│ +└──────────────────────────────────────────────────────────────┘ +``` + +| Information shown | Source | +|---|---| +| Headline, subcopy, reassurance line | Static copy | +| Three-step explainer | Static copy | + +**Actions:** type an address → submit. +**Next:** B, or A1 / A2 below. + +--- + +### A1 · Bad address (inline, does not leave the page) + +``` + ┌────────────────────────────────┐ ┌──────────────────┐ + │ 127.0.0.1 │ │ Read my site → │ + └────────────────────────────────┘ └──────────────────┘ + ⚠ Enter a domain name, not an IP address. +``` + +One line, under the field, in the visitor's language. Never a raw error. +The messages come from `lib/crawl/validate-url.ts`: + +| Situation | Message | +|---|---| +| Empty | Enter a website address. | +| Not parseable | That doesn't look like a website address. | +| IP address (v4 or v6) | Enter a domain name, not an IP address. | +| `localhost`, `.local`, `.internal` | That address is not reachable from the public internet. | +| Single word, no dot | Enter a full domain, like example.com. | +| `ftp:`, `file:`, `javascript:` | Only http and https addresses can be read. | +| Credentials in the address | Addresses with login details are not accepted. | +| Unusual port | Only standard web ports can be read. | + +--- + +### A2 · Over the daily limit + +``` +┌──────────────────────────────────────────────────────────────┐ +│ │ +│ You've used your three checks for today │ +│ │ +│ The tool is free and we cap it so it stays that way. │ +│ Your checks reset at midnight UTC. │ +│ │ +│ ┌──────────────────────┐ │ +│ │ Talk to us instead →│ │ +│ └──────────────────────┘ │ +└──────────────────────────────────────────────────────────────┘ +``` + +Two different caps land here, with different copy: + +- **3 sessions per visitor per day** → "your three checks for today" +- **200 messages per day across everyone** → "the tool is busy today, try tomorrow" + +Neither is an error. Both offer the contact route, because someone hitting the +cap is interested. + +--- + +## 3. B · Reading the site + +``` +┌──────────────────────────────────────────────────────────────┐ +│ northwindlogistics.com │ +│ │ +│ Reading your site │ +│ │ +│ 12 of 20 pages about 12s left │ +│ [██████████████████░░░░░░░░░░░░░░░░░░░░] │ +│ │ +│ ✓ / │ +│ ✓ /about │ +│ ✓ /services │ +│ ✓ /services/freight │ +│ ✓ /contact │ +│ ▸ /industries reading… │ +│ │ +│ Found your sitemap. Reading the pages a buyer would │ +│ land on first. │ +└──────────────────────────────────────────────────────────────┘ +``` + +| Information shown | Source | +|---|---| +| Host being read | The submitted URL | +| Page count + progress | Streamed, one event per page fetched | +| Page paths, as they land | Streamed live | +| Status line | Changes: "Looking for your sitemap" → "Found your sitemap" → "No sitemap, following your links" | + +**Why the list and not a spinner:** the page paths are proof we are really +reading their site, and it is the moment the visitor first believes the tool. +A spinner is indistinguishable from a hang. + +**Actions:** none. It cannot be cancelled; it takes under 20 seconds by design. +**Next:** C if text was found, E if not, B1 if the site could not be reached. + +--- + +### B1 · Cannot reach it + +``` +┌──────────────────────────────────────────────────────────────┐ +│ We couldn't reach northwindlogistics.com │ +│ │ +│ What we tried │ +│ ─────────────────────────────────────────────────────── │ +│ robots.txt Blocked us from every page │ +│ Homepage No response after 20 seconds │ +│ │ +│ Some sites block automated readers. Search engines and │ +│ AI assistants hit the same wall we just did. │ +│ │ +│ ┌────────────────┐ ┌──────────────────┐ │ +│ │ Try another → │ │ Talk to us → │ │ +│ └────────────────┘ └──────────────────┘ │ +└──────────────────────────────────────────────────────────────┘ +``` + +Names the specific blocker: robots disallow, timeout, DNS failure, or a +non-200. Never "something went wrong". + +--- + +## 4. C · Chat + +``` +┌──────────────────────────────────────────────────────────────┐ +│ CHATTING WITH ┌──────────────┐ │ +│ northwindlogistics.com │ 5 of 8 left │ │ +│ ────────────────────────────────────────────────────────── │ +│ │ +│ ┌────────────────────────────────────────┐ │ +│ │ I've read 17 pages of northwind… │ ← bot │ +│ │ Ask me anything a customer might ask. │ │ +│ └────────────────────────────────────────┘ │ +│ │ +│ ┌─────────────────────────────────────────┐ │ +│ you → │ What exactly does this company do? │ │ +│ └─────────────────────────────────────────┘ │ +│ │ +│ ┌────────────────────────────────────────┐ │ +│ │ Northwind provides freight forwarding │ ← bot, answers │ +│ │ and warehousing across the UK… │ │ +│ └────────────────────────────────────────┘ │ +│ │ +│ ┌─────────────────────────────────────────┐ │ +│ you → │ What does it cost? │ │ +│ └─────────────────────────────────────────┘ │ +│ │ +│ ┌────────────────────────────────────────┐ │ +│ │ ! NOT ON THE SITE │ ← bot, honest │ +│ │ I couldn't find that. There's no │ miss. THIS is │ +│ │ pricing page, and the services pages │ the product. │ +│ │ describe what's offered without │ │ +│ │ naming a price or how a quote works. │ │ +│ └────────────────────────────────────────┘ │ +│ │ +│ TRY ASKING │ +│ ( What exactly does this company do? ) │ +│ ( Who is it for? ) ( What does it cost? ) │ +│ ( Who have they worked with before? ) │ +│ │ +│ ┌────────────────────────────────┐ ┌──────────────────┐ │ +│ │ Ask something a customer would │ │ Send → │ │ +│ └────────────────────────────────┘ └──────────────────┘ │ +│ │ +│ [ Skip to my report → ] │ +└──────────────────────────────────────────────────────────────┘ +``` + +| Information shown | Source | +|---|---| +| Host + pages read | Session | +| Messages remaining | `8 − messages_used` | +| Bot answers | Model, from the crawled text only | +| "Not on the site" marker | Set when the bot cannot answer — the raw material of the report | +| Suggested questions | 4 of the 10 scored buyer questions | + +**Actions:** type a question, click a suggested question, or skip to the report. + +**Two behaviours that matter:** +- The visitor can leave for the report at any time. Never trap them at 8. +- If the model is busy, the bot shows *"one moment…"* and retries down the free + model chain. It never shows a raw error mid-conversation. + +**Next:** D, on the 8th message or on "skip to my report". + +--- + +## 5. D · The report + +``` +┌──────────────────────────────────────────────────────────────┐ +│ northwindlogistics.com │ +│ │ +│ 4 of 10 │ +│ buyer questions your website can answer │ +│ │ +│ The bot was limited by what your site says, not by the │ +│ bot. Here is what it could not find. │ +│ │ +│ ✗ What does it cost, or how is pricing decided? │ +│ ✗ How long does a typical project take? │ +│ ✗ What results have they actually produced? │ +│ ✗ What makes them different from the alternatives? │ +│ ✗ Who actually does the work? │ +│ ✗ What happens between contact and finished work? │ +│ │ +├──────────────────────────────────────────────────────────────┤ +│ THE REST OF THE REPORT │ +│ │ +│ Coverage 3 of 5 page types │ +│ · Missing: pricing or process │ +│ · Missing: proof — no case studies or results │ +│ │ +│ Structure Weak │ +│ · 9 of 17 pages have no H2 at all │ +│ · /services is one block with 14 styled divs │ +│ │ +│ Crawlability Passes │ +│ · Sitemap found at /sitemap.xml │ +│ · robots.txt allows crawling │ +│ · Text readable without JavaScript │ +│ │ +│ Specificity 11 vague claims │ +│ · "world-class service" — /about │ +│ · "industry-leading turnaround" — /services, no number │ +│ │ +├──────────────────────────────────────────────────────────────┤ +│ This is a content problem, not a bot problem. │ +│ │ +│ Every question above is one a buyer asks before they │ +│ get in touch. We rebuild sites so the answers are on │ +│ the page. │ +│ │ +│ ┌──────────────────────┐ ┌────────────────────────────┐ │ +│ │ Talk about a rebuild │ │ Want this bot on your site?│ │ +│ └──────────────────────┘ └────────────────────────────┘ │ +│ │ +│ Fixed something? Read my site again │ +└──────────────────────────────────────────────────────────────┘ +``` + +| Score | What it says | Evidence it shows | +|---|---|---| +| **Answerability** | "4 of 10" — the headline | The exact questions that failed | +| Coverage | How many of 5 page types exist | Names the missing ones | +| Structure | Are headings real and nested | Heading depth per page | +| Crawlability | Sitemap, robots, text without JS | Pass/fail per check + the blocker | +| Specificity | Concrete facts vs vague copy | Quotes their own words | + +**Every number points at something.** No score appears without the evidence +underneath it. + +**Actions:** rebuild CTA · email capture · re-crawl. +**Next:** `/contact`, F, or back to B. + +--- + +### D1 · When the site scores well + +``` +│ 9 of 10 │ +│ buyer questions your website can answer │ +│ │ +│ Your site answers almost everything a buyer asks. That │ +│ is rare. The one gap: │ +│ │ +│ ✗ What does it cost, or how is pricing decided? │ +``` + +A high scorer is the most interesting visitor on the page and the current +draft has nothing gracious to say to them. The report must not read as a +failure notice when the site is good — same layout, different framing, and +the CTA shifts from "this is broken" to "you're most of the way there". + +**Open question for copy.** Flagged, not solved. + +--- + +## 6. E · Unreadable site + +``` +┌──────────────────────────────────────────────────────────────┐ +│ northwind-app.io │ +│ │ +│ We could not read your site │ +│ │ +│ We reached your pages, but they returned almost no │ +│ readable text. Everything is drawn by JavaScript after │ +│ the page loads. An AI assistant reading your site sees │ +│ what we saw: an empty page. │ +│ │ +│ WHAT WE FOUND │ +│ ───────────────────────────────────────────────────────── │ +│ Sitemap Found — 14 URLs │ +│ robots.txt Allows crawling │ +│ Readable text without JavaScript 38 words / 14 pages │ +│ │ +│ There is nothing to chat with, and that is the finding. │ +│ Search engines, AI assistants and previews all read a │ +│ page the way we just did. │ +│ │ +│ ┌──────────────────────┐ ┌────────────────────────────┐ │ +│ │ Talk about a rebuild │ │ Read my site again → │ │ +│ └──────────────────────┘ └────────────────────────────┘ │ +└──────────────────────────────────────────────────────────────┘ +``` + +**No chat is offered.** A bot with nothing to say makes EPYC look broken; this +screen makes their site look broken, which is both true and the point. + +--- + +## 7. F · Email capture + +``` +┌──────────────────────────────────────────────────────────────┐ +│ Want this bot on your site? │ +│ │ +│ We're building an embeddable version. Free, unlimited │ +│ messages, runs on your own site. Leave your email and │ +│ we'll tell you when it's ready. │ +│ │ +│ ┌────────────────────────────────┐ ┌──────────────────┐ │ +│ │ you@company.com │ │ Notify me → │ │ +│ └────────────────────────────────┘ └──────────────────┘ │ +│ │ +│ ── after submit ──────────────────────────────────────── │ +│ ✓ We'll be in touch. Your report stays on this page. │ +└──────────────────────────────────────────────────────────────┘ +``` + +This capture is the phase-two gate. Its click count decides whether the +embeddable widget gets built at all, so it must be a real capture and not a +dead button. + +--- + +## 8. What the visitor never sees + +Deliberate omissions, listed so nobody adds them back by accident: + +- **No account, no signup, no password.** Email is asked for once, at the end, optional. +- **No model name.** Not "powered by Nemotron". It is a diagnostic, not an AI product. +- **No raw errors, no status codes, no stack traces.** Every failure names a cause in plain words. +- **No critique from the bot.** It stays in character as their support assistant. All criticism lives in the report. +- **No score before the chat.** The conversation has to come first or the score is a claim instead of a recap. +- **No paywall, no "upgrade to see the rest".** The whole report is free. + +--- + +## 9. Still open + +1. **D1 copy — the good-site framing.** Layout is settled, wording is not. +2. **Re-crawl wait.** Re-crawling returns to B for another 20 seconds. Acceptable, or does it need a lighter treatment when only one page changed? +3. **Mid-chat model exhaustion.** If every free tier is rate-limited at once, the "one moment" state has to end somewhere. Offer the report early, or hold? diff --git a/docs/ai-chatbot-plan.md b/docs/ai-chatbot-plan.md new file mode 100644 index 0000000..8cfe949 --- /dev/null +++ b/docs/ai-chatbot-plan.md @@ -0,0 +1,130 @@ +# AI Chatbot — Execution Plan (Phase One) + +**Status:** for approval · **Date:** 14 August 2026 · **Route:** `/tools/ai-chatbot` +**Source spec:** *EPYC — AI Chatbot Demo* (decisions + reasoning) +**Companion docs:** [`ai-chatbot-architecture.md`](./ai-chatbot-architecture.md) (why the system is shaped this way) · [`ai-chatbot-tech.md`](./ai-chatbot-tech.md) (route contracts, schema, build reference) + +--- + +## What we use + +Everything runs in the existing website repo on Cloudflare. No new infrastructure. + +| Piece | Role | +|---|---| +| Workers | the app and the API | +| D1 | sessions, page text, daily counters, captured emails | +| OpenRouter | `nemotron-3.5-lightning:free`, falling back to `nemotron-3-super:free` then `nemotron-3-ultra:free` — free at every tier | +| AI SDK | streaming chat, with the UI restyled to our design system | + +**Verified 14 August 2026 against OpenRouter's live model list:** Lightning's free variant is $0 in / $0 out with a **1,000,000-token context window**. Roughly 20 pages of site text is about 80,000 tokens, so the whole site fits in one request with 12× headroom. This was the assumption the no-search-index decision rested on; it now checks out. + +**The fallback is free too.** When the free tier is busy we step down to Nemotron 3 Super (free, same 1M window) and then Nemotron 3 Ultra (free, 512K — still six times what we need). All three are $0. A paid model stays configured but switched off, as break-glass only. + +**Phase one therefore has no per-message cost at all.** One caveat to test: OpenRouter's free tiers may share a single account-wide quota rather than one per model, in which case stepping down between them buys nothing and the paid break-glass becomes real. Cheap to check, and listed below. + +No search index, no embeddings, no second server, no job queue. At 20 pages the site text fits in one request, so retrieval would add moving parts without improving answers. + +Three calls worth flagging: + +1. **Lightning free over Nemotron 3 Ultra** — 1.2s vs 6s per answer, and free. +2. **Crawling inside the app** — live progress instead of a background queue. +3. **Quiet scoring** — the report is scored after message one so it appears instantly at the end. + +--- + +## How it works + +1. Check the URL and the daily limits, read `robots.txt` and the sitemap, fetch up to 20 pages, extract the text. Progress streams to the screen. +2. Site text goes into the model's context. Eight messages. +3. Show the report. +4. Two buttons: talk about a rebuild, or "want this on your site?" which captures an email. + +--- + +## The report + +Five scores, all from the spec, each backed by evidence from the visitor's own site. No number we cannot point at. + +| Score | What it measures | Evidence shown | +|---|---|---| +| **Answerability** | How many of 10 standard buyer questions the site can answer | "4 of 10" plus the list it could not answer. **The headline number.** | +| **Coverage** | Whether pages exist for what they do, who they serve, pricing or process, proof, and contact | Names the missing page types | +| **Structure** | Whether headings are real and nested, or the page is one large block of markup | Heading depth per page | +| **Crawlability** | Sitemap present, robots not blocking, readable text without JavaScript | Pass or fail per check, naming the specific blocker | +| **Specificity** | Concrete facts against vague marketing copy | Quotes examples from their own pages | + +Structure, Crawlability and Specificity are measured directly from what we crawled and cost nothing. Answerability and Coverage need judgement, so **one** model call answers both at once rather than eleven separate calls. If that call fails we still show the other three rather than an error. + +**Which model scores it — now settled, for free.** Lightning is small and fast, which is right for the chat but thin for judgement across a whole site. Since the scoring call runs once per session rather than eight times, it runs on Nemotron 3 Super's free variant instead: a larger model, same 1M window, still $0. Compare the two across ten real sites before launch and keep whichever reads better. + +Specificity is the borderline one. A phrase list gets most of the way, but if the quoted examples come out weak in testing it moves into the model call. Same call, one extra field, no extra cost. Decide after seeing real output. + +--- + +## Phase one — order of work + +| # | Step | Scope | Done when | +|---|---|---|---| +| 1 | Write the 10 buyer questions | needs approval, blocks step 5 | List agreed and committed to `data/` | +| 2 | Foundations | tables, URL safety, counter, 2 tests | Counter works on a fresh day and stops at its limit; validator rejects internal addresses | +| 3 | Reading websites | sitemap, fetching, extraction, progress | A real prospect site returns 15–20 pages of clean text | +| 4 | The chat | prompt, streaming, 8 message cap | The bot answers from the site and admits clearly when it cannot | +| 5 | The report | the five scores above | Two different sites produce visibly different, defensible scores | +| 6 | The page | largest step — five screens, mostly UI | Matches the approved wireframe, passes design review, works at 375px | +| 7 | Tracking and email capture | funnel events, drop-off, interest button | One session produces a clean event trail from paste to click | +| 8 | Launch | migrations, secrets, sitemap, privacy line | Staging runs a full session against a real site | + +Steps 3 to 6 are the bulk of the work. Each step is reviewable on its own. + +**The design is settled ahead of step 6.** A click-through wireframe of all five screens is built at `/tools/ai-chatbot` using the real design system, and the screen inventory, component list, and copy rules are written up in [`ai-chatbot-tech.md` → Design](./ai-chatbot-tech.md#design). Step 6 is then wiring, not deciding. + +--- + +## Phase two — later, gated on the email count + +| Item | Note | +|---|---| +| Embeddable widget | Keys tied to a domain, per-customer limits, terms | +| Nightly re-crawl | Content fingerprint, only re-reads changed pages, IDs stay stable so embed codes never break | +| Browser-based crawler | So JavaScript-heavy sites read correctly | +| Search-based retrieval | Object storage plus indexing, once many bots run all day and sending the whole site each time gets expensive | + +Manual re-crawl ships in phase one as a button. Nothing in phase one needs undoing to reach any of this. + +--- + +## Limits + +20 pages per site · 500 KB per page · 20 second crawl · 8 messages per session · 3 sessions per visitor per day · 200 messages a day across everyone. + +With every tier on a free model, these control rate limiting and abuse rather than cost — there is no per-message bill to protect. The caps exist to stop someone pointing a script at us and to stay inside OpenRouter's free-tier limits. 200/day is about 25 full demos; if the tool works, this is the first number we raise. + +These limits are enforced in the app, not by convention: the crawler caps pages, bytes, depth, redirects and total time; the message route caps messages per session; and a single atomic database statement caps sessions per visitor per day and messages per day across everyone. A request that would exceed a cap is rejected before any model call is made. + +--- + +## Decisions needed + +1. **Approve demo now** — widget gated on the email button. +2. **Approve the 10 buyer questions** once drafted. +3. **Owner for the privacy line.** We read other people's sites, store that text, keep transcripts and collect emails. Launch blocker, not an engineering task. +4. **Confirm 200 messages a day** to start. + +--- + +## Checks first + +**Resolved (14 August 2026).** + +- ~~Lightning's context window fits ~20 pages in one request.~~ Confirmed: 1,000,000 tokens against an ~80,000-token corpus. +- ~~Exact OpenRouter model IDs.~~ Confirmed: `nvidia/nemotron-3.5-lightning:free` and `nvidia/nemotron-3.5-lightning`. +- ~~Whether the AI SDK and the OpenRouter provider conflict with anything installed.~~ Confirmed compatible: `ai@7.0.65` + `@openrouter/ai-sdk-provider@3.0.0`, whose zod requirement is already satisfied by the version in this repo. + +**Still open, and each is small.** + +1. ~~**Whether the three free models share one quota.**~~ **Answered, and the answer is bad: yes.** OpenRouter meters free usage per *account* (`free-models-per-day`), so stepping from Lightning to Super to Ultra buys nothing — we hit it in testing and all three refused at once. + + Without credits the account gets about **50 free requests a day**, against a planned cap of 200 messages a day. **Adding $10 of credits raises it to 1,000 free requests a day** and still costs nothing per message. That $10 is a launch prerequisite. It is the cheapest item on this entire list and currently the one that stops the tool working. +2. **Live progress needs a 30-minute spike first.** The crawl streams progress to the screen through our Cloudflare deployment adapter. That should work, and the whole "read the site in the app" decision assumes it. Confirm it with a throwaway route before step 3 depends on it. If it turns out responses are buffered, the fallback is a plain progress bar — the architecture does not change, only the polish. +3. ~~**CI never runs on our pull requests.**~~ **Done** — it now triggers on `main` and `production`, runs the tests, and is on a current Node version. diff --git a/docs/ai-chatbot-tech.md b/docs/ai-chatbot-tech.md new file mode 100644 index 0000000..4036f1a --- /dev/null +++ b/docs/ai-chatbot-tech.md @@ -0,0 +1,443 @@ +# AI Chatbot — Technical Reference + +**Companion docs:** [`ai-chatbot-plan.md`](./ai-chatbot-plan.md) (scope, order of work, approvals) · [`ai-chatbot-architecture.md`](./ai-chatbot-architecture.md) (system shape and the reasoning behind it) + +Build reference for phase one: file layout, route contracts, schema, crawler rules, prompts, scoring, caps, dependencies, secrets and the deploy checklist. Read `DESIGN.md` before writing any UI, and `AGENTS.md` before touching routing conventions. + +--- + +## File layout + +``` +app/(my-app)/tools/ai-chatbot/page.tsx landing + tool UI (client island inside a server page) +app/api/tools/chatbot/crawl/route.ts POST — inline crawl, SSE progress +app/api/tools/chatbot/message/route.ts POST — chat turn, SSE tokens +app/api/tools/chatbot/diagnosis/route.ts GET — stored report JSON +app/api/tools/chatbot/interest/route.ts POST — email capture (phase-two gate) + +components/sections/chatbot-tool.tsx the tool: URL box → progress → chat → report +components/ui/chat-*.tsx thread, bubble, composer — restyled to DESIGN.md tokens + +lib/crawl/validate-url.ts URL safety (tested) +lib/crawl/sitemap.ts robots.txt Sitemap: → sitemap index → page URLs +lib/crawl/fetch-pages.ts polite concurrent fetch, byte + time caps +lib/crawl/extract.ts HTML → text + structure metadata +lib/tools/counters.ts atomic daily counters (tested) +lib/tools/diagnosis.ts three deterministic scores + one model call +lib/tools/prompt.ts system prompt + corpus assembly +data/buyer-questions.ts the 10 questions — editable, approval-gated + +db/migrations/0003_tool_sessions.sql +``` + +`lib/crawl/` is written as a standalone module with no route coupling: the Website Grader and llms.txt Generator are its next two consumers. + +--- + +## D1 schema + +New migration `db/migrations/0003_tool_sessions.sql`, same `DB` binding as the contact form. SQLite, idempotent, applied manually per repo convention (see [Deploy checklist](#deploy-checklist)). + +```sql +-- One row per demo session. +CREATE TABLE IF NOT EXISTS tool_sessions ( + id TEXT PRIMARY KEY, -- opaque, client-visible + tool TEXT NOT NULL, -- 'chatbot' | 'grader' | 'llms-txt' + target_url TEXT NOT NULL, + host TEXT NOT NULL, -- normalised, drives the 24h crawl reuse + ip_hash TEXT NOT NULL, -- HMAC-SHA256(ip, TOOLS_IP_SALT), never a bare hash + status TEXT NOT NULL, -- 'ready' | 'empty' | 'failed' + pages_crawled INTEGER NOT NULL DEFAULT 0, + messages_used INTEGER NOT NULL DEFAULT 0, + diagnosis_json TEXT, -- scored once, after message 1 + transcript_json TEXT, -- what visitors actually ask + email TEXT, -- set only on interest capture + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +-- Extracted corpus, one row per page. +CREATE TABLE IF NOT EXISTS tool_pages ( + session_id TEXT NOT NULL REFERENCES tool_sessions(id), + url TEXT NOT NULL, + title TEXT, + text TEXT NOT NULL, + meta_json TEXT, -- heading depths, word count, JS-empty flag + PRIMARY KEY (session_id, url) +); + +-- Atomic daily counters. key: 'global-messages' | 'ip:' | later 'embed:' +CREATE TABLE IF NOT EXISTS tool_counters ( + day TEXT NOT NULL, + key TEXT NOT NULL, + n INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (day, key) +); + +-- Email captures. kind: 'embed' | 'model:' +CREATE TABLE IF NOT EXISTS tool_interest ( + id TEXT PRIMARY KEY, + session_id TEXT REFERENCES tool_sessions(id), + kind TEXT NOT NULL, + email TEXT, + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +CREATE INDEX IF NOT EXISTS idx_tool_sessions_host_created + ON tool_sessions (host, created_at DESC); -- 24h crawl reuse lookup +``` + +`tool_embeds` is **not** in this migration. It ships with the widget in phase two. + +### The counter statement + +One statement, no read-then-increment, no Durable Object. This is the corrected form — the spec's `UPDATE ... WHERE n < ?` affects zero rows when today's row does not exist yet, which reads as "cap hit" on the first request after every midnight. + +```sql +INSERT INTO tool_counters (day, key, n) VALUES (?, ?, 1) +ON CONFLICT(day, key) DO UPDATE SET n = n + 1 WHERE n < ?; +``` + +`meta.changes === 0` means **capped**, and nothing else. `day` is UTC `YYYY-MM-DD`, so the reset never moves with DST. Positional `?` rather than numbered `?1` — matches `app/api/contact/route.ts`. + +**Built:** `lib/tools/counters.ts` — `bumpCounter()` consumes one unit and returns whether it was allowed; `underLimit()` peeks without consuming, for the crawl route's check-before-work; `CAPS` and `counterKeys` hold the numbers and key formats. + +--- + +## Route contracts + +All four routes live under `app/api/tools/chatbot/`. They follow `app/api/contact/route.ts`: parse JSON, zod `safeParse`, 400 on failure with `fieldErrors`, `getCloudflareContext()` for bindings. + +### `POST /api/tools/chatbot/crawl` → `text/event-stream` + +```jsonc +// request +{ "url": "https://example.com" } +``` + +```jsonc +// request — `force` bypasses the 24h reuse cache +{ "url": "https://example.com", "force": false } +``` + +Order of operations — cheapest rejection first: + +1. Validate and normalise the URL (`lib/crawl/validate-url.ts`). Reject non-`http(s)` schemes, IP literals, `localhost`, userinfo in the authority, and hosts that resolve to a private range. `400`. +2. **Read** `ip:` against 3/day. Over cap → `429` with a "back tomorrow" body. Do **not** increment yet — see below. +3. **Reuse:** unless `force` is set, if a `tool_sessions` row for the same `host` is under 24h old and `status = 'ready'`, copy its `tool_pages` into a new session and stream a single `done` event. Instant, and it stops us re-crawling a prospect every time sales demos the same domain. +4. Otherwise crawl (see [Crawler rules](#crawler-rules)), streaming one event per page. +5. Write `tool_sessions` + `tool_pages`. If total extracted text is under the empty-corpus threshold, set `status = 'empty'` and score the report now (see [Report scoring](#report-scoring)). +6. **Increment `ip:` only once a session actually exists.** Checking and incrementing separately is not atomic, but the failure mode is one extra free crawl under a race, which is the right way to be wrong here — incrementing up front means a typo'd URL or a dead host burns one of the visitor's three daily sessions. + +**`force: true` is what the manual re-crawl button sends.** Without it the button collides with the 24h reuse cache: a visitor who reads the report, fixes their content and re-runs would be handed the stale corpus and an unchanged score, which breaks the exact conversion moment the button exists to create. Re-crawls still consume a session against `ip:`. + +The IP comes from the `CF-Connecting-IP` header, then HMAC-SHA256 with `TOOLS_IP_SALT`. Never store or log the raw address. + +SSE events: + +``` +event: page data: {"url":"/pricing","title":"Pricing","index":4,"total":20} +event: done data: {"sessionId":"...","pages":17,"status":"ready"} +event: error data: {"message":"..."} +``` + +`status: "empty"` means the client skips the chat entirely and goes straight to the report with Crawlability failed and the specific blocker named. A bot with nothing to say makes EPYC look broken; the report makes their site look broken, which is the point. + +### `POST /api/tools/chatbot/message` → `text/event-stream` + +```jsonc +{ "sessionId": "...", "message": "what does it cost?" } +``` + +1. Load session. `messages_used >= 8` → `409` with `{ capped: true }`; the client shows the report. +2. Bump `global-messages` against 200/day. Over → `429`, "back tomorrow" state. +3. Assemble corpus + history (see [Prompt](#prompt)), stream from OpenRouter via the AI SDK. +4. Increment `messages_used`, append to `transcript_json`. +5. **On message 1 only:** `waitUntil(scoreDiagnosis(sessionId))` — the report is computed while the visitor is still typing. + +Rate-limit handling: on `429` from the free tier, retry once with jitter, then fall back to the paid model for that turn. A rate limit degrades quality, never the page. + +### `GET /api/tools/chatbot/diagnosis?sessionId=…` + +Returns `diagnosis_json`. If the background scoring has not landed yet, compute the three deterministic dimensions inline and return them with `"partial": true` — never an error, never a spinner at the moment we ask for the click. + +**Session ids are the only credential on `/message` and `/diagnosis`.** Mint them from `crypto.randomUUID()` — unguessable, and the per-session message cap bounds what holding one gets you. Worth stating so nobody later assumes an auth layer that was never there. + +### `POST /api/tools/chatbot/interest` + +```jsonc +{ "sessionId": "...", "kind": "embed", "email": "someone@example.com" } +``` + +Writes `tool_interest`, mirrors the email onto `tool_sessions.email`. This row is the phase-two gate. Honeypot field, same pattern as the contact form. + +--- + +## Crawler rules + +| Rule | Value | Why | +|---|---|---| +| Page discovery | `robots.txt` `Sitemap:` directive **first**, then `/sitemap.xml` | The directive is the standard discovery path and we are fetching robots anyway | +| Sitemap index | follow **one** level down | `sitemap.xml` is a sitemap *index* on most WordPress/Yoast, Shopify and large sites; naive parsing returns zero pages | +| No sitemap | homepage + same-host internal links, depth 1 | | +| URL preference | shallow paths, and slugs matching `about\|service\|product\|pricing\|contact\|work\|case` | These are the pages the report scores | +| Pages | 20 hard | | +| Bytes per page | 500 KB, then abort that page | | +| Redirects | 3 | | +| Concurrency | 5–6 in flight per host | We are pointing traffic at someone else's server | +| User-Agent | names EPYC + a contact URL | Anonymous scrapers get blocked, permanently, for every future visitor | +| Wall clock | 20s hard stop, proceed with what landed | One slow host must not hang the demo | +| `robots.txt` | honour `Disallow` for the paths crawled | | +| Extraction | semantic containers; drop `nav`, `footer`, `script`, `style` | | +| Truncation | ~4k tokens per page | | + +**Never reflect raw fetched HTML back to the browser.** Extracted text only. + +`lib/crawl/extract.ts` returns the text *and* the structure metadata the report needs in the same pass — heading tags and depth, word count, and whether the page yielded text at all. Crawling twice for that would be silly. + +--- + +## Prompt + +Corpus goes in as one block — page title, URL, extracted text per page — then conversation history. The system prompt does four things: + +1. Answer **only** from the supplied page text. Never use outside knowledge about the company. +2. When the text does not contain the answer, say so plainly and name what is missing. Do not guess, do not pad. +3. Stay in the persona of a support assistant for **that company**. Never mention EPYC. Never break character to critique the site — the critique belongs in the report. +4. Keep answers short. Two or three sentences unless asked for detail. + +Point 2 is load-bearing. The honest "I could not find that on the site" answers are the raw material the report is built from, and they are the sales argument. A bot that bluffs destroys the whole mechanic. + +Set `reasoning_effort` low or off — the bot answers from supplied text rather than solving anything. + +**No prompt caching.** The full corpus resends on every message. On the free tier that costs nothing but latency; it is the reason the paid fallback needs a cheap model rather than a large one. + +--- + +## Report scoring + +`lib/tools/diagnosis.ts` returns one JSON object, stored on the session. Every number is backed by evidence pulled from the crawl — nothing invented, nothing we cannot point at. + +### Deterministic — from crawl metadata, no model call + +| Dimension | Computed from | Evidence emitted | +|---|---|---| +| **Structure** | heading tags and nesting per page, from `tool_pages.meta_json` | heading depth per page; flags "one large block of markup" | +| **Crawlability** | sitemap found?, robots blocking?, text present without JS execution? | pass/fail per check with the specific blocker named | +| **Specificity** | phrase list of vague marketing claims + "claims without numbers" heuristic over page text | quoted examples from their own pages | + +### One model call — Answerability + Coverage + +A single structured JSON response covering all 10 buyer questions **and** the five page types. Not eleven calls. + +```jsonc +{ + "answerability": { + "answered": 4, "total": 10, + "unanswered": ["what does it cost", "how long does it take", "..."] + }, + "coverage": { "missing": ["pricing or process", "proof"] } +} +``` + +Failure of this call renders the three deterministic dimensions with `"partial": true`. Not an error state. + +### When scoring runs + +| Path | Trigger | Model call | +|---|---|---| +| Normal | `waitUntil()` on message 1 | yes | +| Empty corpus (`status = 'empty'`) | end of crawl — there is no message 1 on this path | **no** — Answerability is 0/10 by construction, Coverage is derived from the page list | +| Manual re-crawl | same as the path the new session lands on | as above | + +`waitUntil` comes off `getCloudflareContext().ctx` — the same accessor the contact route uses for bindings. + +### Model choice for this call + +Lightning is a 3B-active/30B MoE: fast, cheap, and well shaped for "answer from the text in front of you", which is the chat. This call is a different job — judgement across an 80k-token corpus, once per session, producing the number we put in front of a prospect. Build it on Lightning, then compare against a larger model over ten real sites before launch. At one call per session the paid tier costs cents; the headline number is not the place to save them. + +**Specificity — decided on real output, 14 Aug 2026.** The plan said to judge this dimension once we had run it against a real site. We did, against epyc.in: + +- **First run: 19 "vague claims", mostly false positives.** The phrase list was scoring blog posts as sales copy — including a post that was itself *mocking* empty language ("ends with a slide that says 'now go be innovative'"), plus article titles from the blog index. +- **Fix: editorial URLs are excluded** (`/blog`, `/news`, `/insights`, `/case-stud`, …). The dimension asks "does this company describe itself concretely", and a blog post is not the company describing itself. Result on the same site: **19 → 1**, and the verdict correctly flipped from *weak* to *pass*. +- **Still imperfect.** The one survivor is the same mocking sentence, on a sales page this time — a phrase list cannot tell a used claim from a criticised one. At one example on a passing verdict that is tolerable. + +**Kept deterministic for now.** The upgrade path is unchanged and still cheap: move it into the scoring call as one extra field. Trigger for doing so is a site where the quotes read as unfair — this dimension is the one most likely to hand an owner an argument, and its whole value is being unarguable. + +The 10 buyer questions live in `data/buyer-questions.ts` as a typed const array so they can be edited without touching logic. They gate step 5 and need sign-off. Three or four of them also render as clickable prompt chips in the composer — that kills the blank-input pause where drop-off happens, and it makes the final score a recap of failures the visitor already watched rather than a claim. + +--- + +## Caps + +| Cap | Value | Enforced in | +|---|---|---| +| Pages per crawl | 20 | crawler | +| Bytes per page | 500 KB, then abort | crawler | +| Extracted text per page | ~4k tokens, truncate | extractor | +| Crawl depth | 1 | crawler | +| Redirects | 3 | fetch options | +| Whole crawl | 20s wall clock | crawl route | +| Messages per session | 8, then report + CTA | message route | +| Sessions per IP per day | 3 | `tool_counters`, key `ip:` | +| Global messages per day | 200 to start | `tool_counters`, key `global-messages` | + +On a free model these bound abuse and upstream rate limits, not spend. No dollar ceiling — spend is bounded structurally. + +--- + +## Dependencies + +| Package | Version verified 14 Aug 2026 | Note | +|---|---|---| +| `ai` | `7.0.65` | peer `zod: ^3.25.76 \|\| ^4.1.8` — satisfied by this repo's `zod@^4.4.3`, no major bump needed | +| `@openrouter/ai-sdk-provider` | `3.0.0` | peer `ai: ^7.0.0` — matches the above. Pin exact, upgrade the pair together | +| `vitest` (dev) | `4.1.10` | **installed.** The repo had no test infra before this | +| `@types/node` (dev) | bumped `^20` → `^22` | `node:sqlite` types. Typecheck stayed clean across the bump | + +### Model chain (OpenRouter, verified 14 Aug 2026) + +Free at every tier. Paid is off by default and exists only as a break-glass flag. + +| Tier | Model ID | Context | Cost | Role | +|---|---|---|---|---| +| 1 — chat | `nvidia/nemotron-3.5-lightning:free` | 1,000,000 | $0 | Primary. ~1.2s per answer | +| 2 — on 429 | `nvidia/nemotron-3-super-120b-a12b:free` | 1,000,000 | $0 | Same window, larger model | +| 3 — on 429 | `nvidia/nemotron-3-ultra-550b-a55b:free` | 512,000 | $0 | Last resort. ~6s, still 6× the corpus size | +| Break-glass | `nvidia/nemotron-3.5-lightning` | 1,000,000 | ~$0.10/M in | Behind `OPENROUTER_ALLOW_PAID=true`. Off by default | +| Scoring call | `nvidia/nemotron-3-super-120b-a12b:free` | 1,000,000 | $0 | See below | + +Implement as an ordered array in `lib/tools/models.ts` — try in order, advance on `429`/`503`, surface the "one moment" state while retrying, and only show a failure after the chain is exhausted. Log which tier served each message so we learn how often tier 1 actually holds. + +**This also settles the scoring-model question at zero cost.** The earlier recommendation was to pay for the Answerability/Coverage call because it is judgement work producing the headline number. Super's free variant is a larger model than Lightning with the same 1M window, and the call runs once per session rather than eight times — so it gets the bigger model for free. No paid tier needed anywhere in phase one. + +### ⚠️ Confirmed 14 Aug 2026: the free cascade does not work + +Tested directly against OpenRouter. All three free tiers return the same error: + +``` +HTTP 429 +"Rate limit exceeded: free-models-per-day. + Add 10 credits to unlock 1000 free model requests per day" +``` + +**The quota is account-wide (`free-models-per-day`), not per-model.** Falling from Lightning to Super to Ultra buys nothing — all three draw on one exhausted bucket. The cascade is worth keeping only for a single model being down, not for rate limits. + +The paid tier cannot rescue it either while the key has no credits: + +``` +HTTP 403 "Key limit exceeded (total limit)" +``` + +**What this means for launch.** Without credits the account gets roughly 50 free requests a day, against a planned global cap of 200 messages a day — so the tool would stop working before lunch. Adding **$10 of credits raises it to 1,000 free requests a day**, which comfortably covers the planned cap and still costs nothing per message. + +That $10 is now a launch prerequisite, not an optimisation. Until it is added, treat every "free tier" statement elsewhere in these docs as conditional on it. + +`@assistant-ui/react` + `@assistant-ui/react-ai-sdk` are **optional and version-coupled to AI SDK majors**. Their selling point is a shadcn theme, which this repo does not have — adopting them means restyling onto `DESIGN.md` tokens, which is real work and is why step 6 is the largest step. Hand-building the thread from `components/ui/` primitives (`Button`, `Textarea`, `Field`) is the lever if step 6 needs cutting: it is one thread, no history, no tool calls, no uploads. + +Never bump `ai` and `@assistant-ui/react-ai-sdk` as a drive-by. Together, deliberately. + +--- + +## Tests + +Two files, 25 cases, all green. Money and security paths only — this is the one place not to be lazy. + +1. **`lib/tools/counters.test.ts`** — allows the first call of a new day (the bug the spec's SQL had), allows exactly `limit` calls then refuses, stays refused, counts keys separately, resets across days, and `underLimit()` consumes nothing. +2. **`lib/crawl/validate-url.test.ts`** — accepts bare domains / subdomains / paths and strips fragments; rejects IPv4 and IPv6 literals, cloud metadata addresses, `localhost` and internal suffixes, single-label hosts, non-web schemes, credentials in the authority, and non-standard ports. + +Run with `pnpm test`. Config is `vitest.config.mts` (`.mts` so Vite's native config loader stops warning). + +**The counter test runs its real SQL against `node:sqlite` rather than `@cloudflare/vitest-pool-workers`.** D1 *is* SQLite, and what is at risk is the semantics of one conditional upsert — `changes` on a cold day versus at the cap — which is engine behaviour, not binding behaviour. One dev dependency instead of a test harness. Move to the workers pool only if we need D1-specific behaviour like `batch()` or session bookmarks. + +Consequence: `node:sqlite` is Node 22+, so **CI's `node-version` moved from 20 to 22** and `@types/node` from `^20` to `^22`. Both are done, and the full typecheck stayed clean across the bump. + +--- + +## Secrets and bindings + +No new bindings. `DB` already exists in `wrangler.jsonc` for both environments. + +| Secret | Set with | +|---|---| +| `OPENROUTER_API_KEY` | `wrangler secret put OPENROUTER_API_KEY --env ` | +| `TOOLS_IP_SALT` | `wrangler secret put TOOLS_IP_SALT --env ` | + +Never in code, never in the client bundle. Regenerate `cloudflare-env.d.ts` with `pnpm cf-typegen` after adding them. + +--- + +## Deploy checklist + +Migrations are **not** auto-applied in this repo. + +```bash +pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --remote \ + --file=db/migrations/0003_tool_sessions.sql +# then the production database, on the production deploy +``` + +- [ ] Migration applied to **both** D1 databases (staging and production names are in `wrangler.jsonc`) +- [ ] `OPENROUTER_API_KEY` and `TOOLS_IP_SALT` set per environment +- [ ] `/tools/ai-chatbot` added to `app/(my-app)/sitemap.ts` +- [ ] Privacy line published and linked from the tool page +- [ ] GA4 events firing: `tool_start`, `tool_crawl_complete`, `tool_message_sent`, `tool_cap_reached`, `tool_diagnosis_viewed` (with the answerability score), `tool_cta_click`, `tool_interest_captured` +- [ ] Clarity recording the drop-off between URL entry and first message — the number that decides whether the tool works + +--- + +## Design + +Reference implementation: `components/sections/chatbot-tool.tsx` — a click-through wireframe of every screen, built from real primitives. Read `DESIGN.md` and `.design-sync/conventions.md` before changing any of it. + +### Screen inventory + +One route, five states. Only one renders at a time. + +| State | Trigger | Ground | Contains | +|---|---|---|---| +| `idle` | first load | `beige` | Eyebrow pill, `text-display` H1, URL field + CTA, reassurance line, three `Disc` steps | +| `crawling` | crawl POST opens | `ink` | `SectionHeading`, progress rail, live page log, "found your sitemap" line | +| `chat` | crawl `done`, `status: ready` | `beige` | Thread header + questions-left `Pill`, bubbles, prompt chips, composer | +| `report` | message cap, or the report link | `ink` → `beige` → `cream` | Answerability headline band, unanswered list, four score rows, two CTAs + re-crawl | +| `empty` | crawl `done`, `status: empty` | `ink` | Blocker named, what-we-found table, rebuild CTA. **No chat offered** | + +The report deliberately spans three tones: the headline number gets its own `ink` band so it reads as the verdict, the detail sits on `beige`, and the CTA lands on `cream`. Do not flatten these into one section — the tone change is what separates "your score" from "what we do about it". + +### Components to graduate + +The wireframe composes three things from tokens because nothing in `components/ui/` was close. On approval they move out of the section file and into `components/ui/`: + +| Wireframe-local | Becomes | Note | +|---|---|---| +| `Bubble` | `components/ui/chat-bubble.tsx` | Three variants: `you` (ink), `bot` (bone), `miss` (crimson-outlined). The `miss` variant is load-bearing — it is the visual the report later refers back to | +| composer row | `components/ui/chat-composer.tsx` | `Field` + `Input` + `Button`, `h-16` to match the field | +| `ScoreRow` | `components/ui/score-row.tsx` | Not a `StatRow` variant. `StatRow` is a 3-up value/label strip for light grounds; this is a full-width row with a verdict and an evidence list | + +### Rules + +- **Tokens only.** No hex, no arbitrary sizes. Type comes from the scale (`text-display`, `text-h2`, `text-body`) which carries family, size, tracking and leading together. +- **Outline buttons on `ink` need `data-on-dark="true"`** or they render ink-on-ink and vanish. +- **`Pill` tone must match the ground** — `cream-on-dark` on ink, `ink-on-light` on beige/cream. +- **Score colour is semantic, not decorative**: `text-crimson` for a failing dimension, `text-ink` for middling, `text-teal-deep` for a pass. Crimson is the CTA colour everywhere else on the site, so use it here only where the finding is genuinely bad. +- **Mobile**: the URL row, composer and CTA pairs all stack (`flex-col sm:flex-row`). The chat thread caps bubbles at `max-w-[85%]`. Test at 375px. +- **Motion**: `Reveal` on entering sections only. The crawl log and streamed tokens are already movement — do not add more. Everything honours `prefers-reduced-motion` via the existing primitives. +- **Accessibility**: the crawl log is `aria-live="polite"`; the questions-left count must be announced, not just coloured; every state change moves focus to the new region's heading. +- **Markdown**: only the rendered state reaches the markdown converter (see `CLAUDE.md` → Markdown for Agents). Before launch, either emit the report as structured data or give the route a builder in `lib/markdown/sources.ts`. + +### Copy direction + +Carried from the source spec, which is the authority on voice. + +- **Voice per `epyc-baseline.md`**: confident, premium, slightly bold. Short sentences. **No em dashes.** Stage-agnostic — never startup-only language. +- **Do not name it as an AI product.** It is a diagnostic that happens to use a chatbot. Headlines sit on the buyer's problem, not the technology: *can a buyer, or an AI assistant, actually find out what this company does from its website?* +- **The bot never breaks character.** It never mentions EPYC and never critiques the site it was built from. All critique lives in the report. +- **The report is evidence, not opinion.** Every line quotes or counts something from their own pages. No adjectives we cannot point at. +- **Proof points come from the marketing repo's `assets/` only.** Accel Atoms AI Chatbot is the relevant credential. Check the `public` flag in `assets/clients.csv` before naming any client on the page. + + + +- ~~**CI never runs on these PRs.**~~ **Fixed.** `.github/workflows/ci.yml` now triggers on `[main, production]`, runs on Node 22, and has a `pnpm test` step. +- **SSE through OpenNext is unproven here.** Nothing in this repo streams a response today. Spike a throwaway route that emits three events over three seconds and confirm the client sees them arrive separately in a deployed preview — not just `next dev`. 30 minutes, before step 3 depends on it. If buffered, degrade to a determinate progress bar; nothing else changes. +- **Subrequest ceiling is plan-dependent** — 1000 on paid Workers plans, 50 on Free. Queues and D1 are already in use, which implies a paid plan, but confirm before the 22-subrequest crawl relies on it. +- **New UI must be checked against existing primitives first.** Per `CLAUDE.md`: run `find components/ui components/sections -name "*.tsx" | sort` and scan `DESIGN.md` §10 and §12 before writing any chat component. `Button`, `Textarea`, `Field`, `Section`, `Container`, `Pill` and `Reveal` all already exist and cover most of this page. +- **A pruning job for unclaimed sessions** is not scheduled by anything today. `tool_pages` for sessions with no interest capture is demo exhaust and should be deleted on a schedule — a Cron Trigger, or a delete-on-write sweep in the crawl route until one exists. diff --git a/lib/crawl/extract.test.ts b/lib/crawl/extract.test.ts new file mode 100644 index 0000000..7e4d8be --- /dev/null +++ b/lib/crawl/extract.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest' +import { EMPTY_WORD_THRESHOLD, extractPage, structureOf } from './extract' + +const page = (body: string, head = 'Northwind Logistics') => + `${head}${body}` + +describe('extractPage', () => { + it('pulls the title', () => { + expect(extractPage(page('

hi

')).title).toBe('Northwind Logistics') + }) + + it('drops script and style content entirely', () => { + const html = page(` + + +

Freight forwarding across the UK.

+ `) + const { text } = extractPage(html) + expect(text).toContain('Freight forwarding') + expect(text).not.toContain('secret') + expect(text).not.toContain('color') + }) + + it('drops repeated chrome so it does not drown the real copy', () => { + const html = page(` + +

We move freight.

+
Copyright 2026 Northwind
+ `) + const { text } = extractPage(html) + expect(text).toBe('We move freight.') + }) + + it('does not fuse words across tag boundaries', () => { + const { text } = extractPage(page('

freight

warehousing

')) + expect(text).toBe('freight warehousing') + }) + + it('decodes entities, named and numeric', () => { + const { text } = extractPage(page('

Bar&Grill — café — 24 hours

')) + expect(text).toBe('Bar&Grill — café — 24 hours') + }) + + it('keeps headings even when they sit inside chrome', () => { + // A heading in a
is still a heading for structure scoring, even + // though the header's text is stripped from the prose. + const { headings } = extractPage(page('

Northwind

Services

')) + expect(headings).toEqual([ + { level: 1, text: 'Northwind' }, + { level: 2, text: 'Services' }, + ]) + }) + + it('truncates to the word budget', () => { + const long = Array.from({ length: 5000 }, () => 'word').join(' ') + const { text, wordCount } = extractPage(page(`

${long}

`), { maxWords: 100 }) + expect(text.split(' ')).toHaveLength(100) + // wordCount reports what was really there, not what we kept. + expect(wordCount).toBe(5000) + }) +}) + +describe('empty detection — the JS-rendered site case', () => { + it('flags a page with almost no text', () => { + const spa = page('
') + expect(extractPage(spa).isEmpty).toBe(true) + }) + + it('does not flag a page with real copy', () => { + const words = Array.from({ length: EMPTY_WORD_THRESHOLD + 10 }, () => 'freight').join(' ') + expect(extractPage(page(`

${words}

`)).isEmpty).toBe(false) + }) + + it('is not fooled by a page that is all script', () => { + const html = page(`
`) + const r = extractPage(html) + expect(r.isEmpty).toBe(true) + expect(r.wordCount).toBeLessThan(EMPTY_WORD_THRESHOLD) + }) +}) + +describe('structureOf', () => { + it('reports nesting when several heading levels are used', () => { + const s = structureOf(extractPage(page('

A

B

C

'))) + expect(s).toEqual({ h1: 1, maxDepth: 3, hasNesting: true }) + }) + + it('reports div soup — no headings at all', () => { + const s = structureOf(extractPage(page('
text
'))) + expect(s).toEqual({ h1: 0, maxDepth: 0, hasNesting: false }) + }) + + it('flags a page with headings but no nesting', () => { + const s = structureOf(extractPage(page('

A

B

'))) + expect(s.hasNesting).toBe(false) + expect(s.h1).toBe(0) + }) +}) diff --git a/lib/crawl/extract.ts b/lib/crawl/extract.ts new file mode 100644 index 0000000..6448ed2 --- /dev/null +++ b/lib/crawl/extract.ts @@ -0,0 +1,131 @@ +/** + * HTML → the text an AI assistant would actually read, plus the structural + * facts the report scores. + * + * One pass produces both. The report needs heading depth and word counts, and + * crawling twice to get them would be silly. + * + * ponytail: string/regex extraction, not a DOM parse. Workers has no + * DOMParser, HTMLRewriter is not available in the Node dev runtime or in + * tests, and pulling in a parser to strip tags would be a dependency for + * something a few replacements do. Known ceiling: malformed nesting can + * over-strip, and content hidden by CSS still counts as text. Upgrade path if + * that ever bites is HTMLRewriter behind this same function signature — no + * caller changes. + */ + +export type Heading = { level: number; text: string } + +export type PageExtract = { + title: string + text: string + headings: Heading[] + wordCount: number + /** True when there is too little text to answer anything — usually a JS-rendered SPA. */ + isEmpty: boolean +} + +/** Below this, a page is not readable content. Drives the `empty` crawl status. */ +export const EMPTY_WORD_THRESHOLD = 50 + +/** Elements whose contents are never page copy. */ +const DROP_CONTENT = /<(script|style|noscript|template|svg|iframe|form|select)\b[^>]*>[\s\S]*?<\/\1>/gi + +/** Chrome that repeats on every page and would drown the real copy. */ +const DROP_CHROME = /<(nav|header|footer|aside)\b[^>]*>[\s\S]*?<\/\1>/gi + +export function extractPage(html: string, opts: { maxWords?: number } = {}): PageExtract { + const maxWords = opts.maxWords ?? 3000 + + const title = decodeEntities( + (html.match(/]*>([\s\S]*?)<\/title>/i)?.[1] ?? '').replace(/\s+/g, ' ').trim(), + ) + + // Drop before anything else, or its text nodes — the above + // all — land in the prose and every page's corpus opens with its own title. + const withoutHead = html.replace(/<head\b[^>]*>[\s\S]*?<\/head>/i, ' ') + + // Headings come from the pre-chrome-stripped body: a heading inside a + // <header> is still a heading for structure-scoring purposes. + const body = withoutHead.replace(DROP_CONTENT, ' ') + const headings = collectHeadings(body) + + const prose = body.replace(DROP_CHROME, ' ') + const text = toText(prose) + const words = text ? text.split(/\s+/) : [] + + return { + title, + text: words.slice(0, maxWords).join(' '), + headings, + wordCount: words.length, + isEmpty: words.length < EMPTY_WORD_THRESHOLD, + } +} + +function collectHeadings(html: string): Heading[] { + const out: Heading[] = [] + for (const m of html.matchAll(/<h([1-6])\b[^>]*>([\s\S]*?)<\/h\1>/gi)) { + const text = toText(m[2]) + if (text) out.push({ level: Number(m[1]), text }) + } + return out +} + +function toText(html: string): string { + return decodeEntities( + html + // Block boundaries become spaces so words don't fuse across tags. + .replace(/<br\s*\/?>/gi, ' ') + .replace(/<\/(p|div|li|tr|h[1-6]|section|article)>/gi, ' ') + .replace(/<[^>]+>/g, ' '), + ) + .replace(/\s+/g, ' ') + .trim() +} + +const NAMED: Record<string, string> = { + amp: '&', + lt: '<', + gt: '>', + quot: '"', + apos: "'", + nbsp: ' ', + mdash: '—', + ndash: '–', + hellip: '…', + rsquo: '’', + lsquo: '‘', + ldquo: '“', + rdquo: '”', +} + +function decodeEntities(s: string): string { + return s.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (whole, body: string) => { + if (body[0] === '#') { + const code = + body[1] === 'x' || body[1] === 'X' + ? Number.parseInt(body.slice(2), 16) + : Number.parseInt(body.slice(1), 10) + return Number.isFinite(code) && code > 0 ? String.fromCodePoint(code) : whole + } + return NAMED[body.toLowerCase()] ?? whole + }) +} + +/** + * Structure signal for the report: are headings real and nested, or is the + * page one undifferentiated block? + */ +export function structureOf(extract: PageExtract): { + h1: number + maxDepth: number + hasNesting: boolean +} { + const levels = extract.headings.map((h) => h.level) + return { + h1: levels.filter((l) => l === 1).length, + maxDepth: levels.length ? Math.max(...levels) : 0, + hasNesting: new Set(levels).size > 1, + } +} diff --git a/lib/crawl/fetch-pages.test.ts b/lib/crawl/fetch-pages.test.ts new file mode 100644 index 0000000..75cdf9d --- /dev/null +++ b/lib/crawl/fetch-pages.test.ts @@ -0,0 +1,181 @@ +import { describe, expect, it } from 'vitest' +import { crawlSite, type CrawlProgress } from './fetch-pages' + +/** + * The orchestration is where the bugs hide — robots before sitemap, sitemap + * index one level down, Disallow actually honoured. A fake fetch lets all of + * that be tested without a network. + */ + +type Routes = Record<string, { body?: string; status?: number; type?: string; location?: string }> + +function fakeFetch(routes: Routes) { + const calls: string[] = [] + const impl = (async (input: RequestInfo | URL) => { + const url = typeof input === 'string' ? input : input.toString() + calls.push(url) + const route = routes[url] + if (!route) return new Response('not found', { status: 404 }) + if (route.location) { + return new Response(null, { status: 301, headers: { location: route.location } }) + } + return new Response(route.body ?? '', { + status: route.status ?? 200, + headers: { 'content-type': route.type ?? 'text/html' }, + }) + }) as unknown as typeof fetch + return { impl, calls } +} + +const html = (body: string, title = 'Page') => + `<!doctype html><html><head><title>${title}${body}` + +/** Enough words that the page is not flagged empty. */ +const copy = (word: string) => `

${Array.from({ length: 80 }, () => word).join(' ')}

` + +describe('crawlSite', () => { + it('takes the sitemap from robots.txt and follows a sitemap index one level', () => { + const { impl, calls } = fakeFetch({ + 'https://example.com/robots.txt': { + body: 'User-agent: *\nDisallow: /admin\nSitemap: https://example.com/sitemap_index.xml', + type: 'text/plain', + }, + 'https://example.com/sitemap_index.xml': { + body: 'https://example.com/pages.xml', + type: 'application/xml', + }, + 'https://example.com/pages.xml': { + body: 'https://example.com/https://example.com/pricing', + type: 'application/xml', + }, + 'https://example.com/': { body: html(copy('home')) }, + 'https://example.com/pricing': { body: html(copy('pricing'), 'Pricing') }, + }) + + return crawlSite('example.com', { fetchImpl: impl }).then((r) => { + expect(r.signals.robotsFound).toBe(true) + expect(r.signals.sitemapFound).toBe(true) + expect(r.pages.map((p) => p.url).sort()).toEqual([ + 'https://example.com/', + 'https://example.com/pricing', + ]) + // The conventional guess is never made when robots names one. + expect(calls).not.toContain('https://example.com/sitemap.xml') + }) + }) + + it('stops immediately when robots disallows everything', async () => { + const { impl, calls } = fakeFetch({ + 'https://example.com/robots.txt': { body: 'User-agent: *\nDisallow: /', type: 'text/plain' }, + 'https://example.com/': { body: html(copy('home')) }, + }) + + const r = await crawlSite('example.com', { fetchImpl: impl }) + expect(r.signals.robotsBlockedAll).toBe(true) + expect(r.pages).toEqual([]) + // and we did not fetch a single page anyway + expect(calls).toEqual(['https://example.com/robots.txt']) + }) + + it('never fetches a disallowed path that the sitemap lists', async () => { + const { impl, calls } = fakeFetch({ + 'https://example.com/robots.txt': { + body: 'User-agent: *\nDisallow: /private\nSitemap: https://example.com/s.xml', + type: 'text/plain', + }, + 'https://example.com/s.xml': { + body: 'https://example.com/https://example.com/private/secret', + type: 'application/xml', + }, + 'https://example.com/': { body: html(copy('home')) }, + 'https://example.com/private/secret': { body: html(copy('secret')) }, + }) + + const r = await crawlSite('example.com', { fetchImpl: impl }) + expect(calls).not.toContain('https://example.com/private/secret') + expect(r.pages).toHaveLength(1) + }) + + it('falls back to following homepage links when there is no sitemap', async () => { + const { impl } = fakeFetch({ + 'https://example.com/': { + body: html(`AboutPricing${copy('home')}`), + }, + 'https://example.com/about': { body: html(copy('about'), 'About') }, + 'https://example.com/pricing': { body: html(copy('pricing'), 'Pricing') }, + }) + + const r = await crawlSite('example.com', { fetchImpl: impl }) + expect(r.signals.sitemapFound).toBe(false) + expect(r.pages.length).toBeGreaterThanOrEqual(2) + }) + + it('reports progress per page, for the live log', async () => { + const { impl } = fakeFetch({ + 'https://example.com/robots.txt': { body: 'Sitemap: https://example.com/s.xml', type: 'text/plain' }, + 'https://example.com/s.xml': { + body: 'https://example.com/', + type: 'application/xml', + }, + 'https://example.com/': { body: html(copy('home'), 'Home') }, + }) + + const events: CrawlProgress[] = [] + await crawlSite('example.com', { fetchImpl: impl, onProgress: (p) => events.push(p) }) + + expect(events.some((e) => e.type === 'status')).toBe(true) + const pageEvents = events.filter((e) => e.type === 'page') + expect(pageEvents).toHaveLength(1) + expect(pageEvents[0]).toMatchObject({ url: 'https://example.com/', title: 'Home', done: 1 }) + }) + + it('flags an unreadable site rather than returning nothing silently', async () => { + const { impl } = fakeFetch({ + 'https://example.com/': { body: html('
') }, + }) + + const r = await crawlSite('example.com', { fetchImpl: impl }) + // The page was reachable but has no readable text — the E screen, not an error. + expect(r.pages.every((p) => p.isEmpty)).toBe(true) + }) + + it('re-validates redirect targets, so a public URL cannot bounce us inward', async () => { + const { impl, calls } = fakeFetch({ + 'https://example.com/robots.txt': { body: 'Sitemap: https://example.com/s.xml', type: 'text/plain' }, + 'https://example.com/s.xml': { + body: 'https://example.com/', + type: 'application/xml', + }, + 'https://example.com/': { location: 'http://169.254.169.254/latest/meta-data/' }, + }) + + const r = await crawlSite('example.com', { fetchImpl: impl }) + expect(calls).not.toContain('http://169.254.169.254/latest/meta-data/') + expect(r.pages).toEqual([]) + }) + + it('rejects an unsafe seed before any request is made', async () => { + const { impl, calls } = fakeFetch({}) + await expect(crawlSite('http://127.0.0.1/', { fetchImpl: impl })).rejects.toThrow(/IP address/i) + expect(calls).toEqual([]) + }) + + it('stops at the wall-clock deadline and keeps what it has', async () => { + const routes: Routes = { + 'https://example.com/robots.txt': { body: 'Sitemap: https://example.com/s.xml', type: 'text/plain' }, + 'https://example.com/s.xml': { + body: `${Array.from({ length: 20 }, (_, i) => `https://example.com/p${i}`).join('')}`, + type: 'application/xml', + }, + } + for (let i = 0; i < 20; i++) routes[`https://example.com/p${i}`] = { body: html(copy('x')) } + + const { impl } = fakeFetch(routes) + // Clock jumps 3s per call — the 20s budget runs out partway through. + let t = 0 + const r = await crawlSite('example.com', { fetchImpl: impl, now: () => (t += 3_000) }) + + expect(r.signals.hitDeadline).toBe(true) + expect(r.pages.length).toBeLessThan(20) + }) +}) diff --git a/lib/crawl/fetch-pages.ts b/lib/crawl/fetch-pages.ts new file mode 100644 index 0000000..49d682f --- /dev/null +++ b/lib/crawl/fetch-pages.ts @@ -0,0 +1,298 @@ +/** + * The crawl itself: robots → sitemap → pages → text. + * + * Runs inline in the route handler rather than in a queue and a second worker, + * so `global_fetch_strictly_public` (set in wrangler.jsonc) applies to every + * fetch here. See docs/ai-chatbot-architecture.md §2.2. + * + * Every limit in docs/ai-chatbot-plan.md is enforced in this file except the + * per-visitor and global daily counts, which are checked by the route before + * it calls in. + */ + +import { extractPage, type PageExtract } from './extract' +import { + fallbackSitemapUrls, + isAllowed, + parseRobots, + parseSitemapXml, + rankUrls, + type RobotsRules, +} from './sitemap' +import { validateUrl } from './validate-url' + +export const LIMITS = { + /** Pages read per site. */ + maxPages: 20, + /** Per page, then we stop reading that response. */ + maxBytes: 500_000, + /** Whole crawl. We return what we have rather than hanging on one slow host. */ + deadlineMs: 20_000, + /** Per single request. */ + requestTimeoutMs: 8_000, + /** + * Sitemaps get longer, because they are worth more: one sitemap yields up to + * 20 ranked URLs, where failing over to link-following yields whatever the + * homepage happens to link to. Measured against real sites — epyc.in's own + * sitemap is server-rendered from a CMS and answers in 0.6s warm but over + * 20s cold, and that variance is normal for CMS-driven sitemaps. + */ + sitemapTimeoutMs: 12_000, + /** In flight at once against one host. We are a guest on their server. */ + concurrency: 5, + /** Redirect hops followed, each re-validated. */ + maxRedirects: 3, + /** Words kept per page. */ + maxWordsPerPage: 3_000, +} as const + +/** + * Identifies us and gives them somewhere to complain. An anonymous scraper + * gets blocked, and once blocked it is blocked for every future visitor. + */ +const USER_AGENT = 'EPYCBot/1.0 (+https://epyc.in/tools/ai-chatbot)' + +export type CrawledPage = PageExtract & { url: string } + +export type CrawlProgress = + | { type: 'status'; message: string } + | { type: 'page'; url: string; title: string; done: number; total: number } + +export type CrawlResult = { + pages: CrawledPage[] + /** Report inputs that are only knowable during the crawl. */ + signals: { + robotsFound: boolean + robotsBlockedAll: boolean + sitemapFound: boolean + /** True when we never got a usable response from the host at all. */ + unreachable: boolean + /** Set when the wall-clock deadline cut the crawl short. */ + hitDeadline: boolean + } +} + +type CrawlOptions = { + onProgress?: (p: CrawlProgress) => void + /** Injectable for tests. */ + fetchImpl?: typeof fetch + now?: () => number +} + +export async function crawlSite(seed: string, opts: CrawlOptions = {}): Promise { + const doFetch = opts.fetchImpl ?? fetch + const now = opts.now ?? (() => Date.now()) + const started = now() + const timeLeft = () => LIMITS.deadlineMs - (now() - started) + const report = (p: CrawlProgress) => opts.onProgress?.(p) + + const checked = validateUrl(seed) + if (!checked.ok) throw new Error(checked.reason) + const origin = new URL(checked.url).origin + + const signals: CrawlResult['signals'] = { + robotsFound: false, + robotsBlockedAll: false, + sitemapFound: false, + unreachable: false, + hitDeadline: false, + } + + // 1. robots.txt — both for permission and for where the sitemap lives. + report({ type: 'status', message: 'Checking what we’re allowed to read' }) + let rules: RobotsRules = { sitemaps: [], disallow: [] } + const robotsBody = await fetchText(new URL('/robots.txt', origin).toString(), doFetch, timeLeft()) + if (robotsBody !== null) { + signals.robotsFound = true + rules = parseRobots(robotsBody) + signals.robotsBlockedAll = !isAllowed('/', rules) + } + if (signals.robotsBlockedAll) return { pages: [], signals } + + // 2. Sitemap — robots' directive first, then the conventional locations. + report({ type: 'status', message: 'Looking for your sitemap' }) + const candidates = rules.sitemaps.length ? rules.sitemaps : fallbackSitemapUrls(origin) + let pageUrls = await collectFromSitemaps(candidates, origin, doFetch, timeLeft) + signals.sitemapFound = pageUrls.length > 0 + + // 3. No sitemap — follow the homepage's own links, one level deep. + if (!signals.sitemapFound) { + report({ type: 'status', message: 'No sitemap. Following your links instead' }) + const home = await fetchText(checked.url, doFetch, timeLeft()) + if (home === null) { + signals.unreachable = true + return { pages: [], signals } + } + pageUrls = rankUrls([checked.url, ...linksIn(home, origin)], origin, LIMITS.maxPages) + } else { + report({ type: 'status', message: 'Found your sitemap' }) + } + + // Never fetch what robots told us not to. + const allowed = pageUrls.filter((u) => isAllowed(new URL(u).pathname, rules)) + const queue = allowed.slice(0, LIMITS.maxPages) + const total = queue.length + const pages: CrawledPage[] = [] + + // 4. Fetch, a few at a time, until the queue empties or time runs out. + let cursor = 0 + const worker = async () => { + while (cursor < queue.length) { + if (timeLeft() <= 0) { + signals.hitDeadline = true + return + } + const url = queue[cursor++] + const html = await fetchText(url, doFetch, Math.min(LIMITS.requestTimeoutMs, timeLeft())) + if (html === null) continue + + const extracted = extractPage(html, { maxWords: LIMITS.maxWordsPerPage }) + pages.push({ ...extracted, url }) + report({ type: 'page', url, title: extracted.title, done: pages.length, total }) + } + } + + await Promise.all( + Array.from({ length: Math.min(LIMITS.concurrency, queue.length) }, () => worker()), + ) + + // The deadline can also be blown by the discovery phase before any page is + // queued — a slow sitemap is the usual culprit. Checking only inside the + // worker loop misses that, and the route would report a clean run that took + // longer than the cap it promises. + if (timeLeft() <= 0) signals.hitDeadline = true + if (pages.length === 0) signals.unreachable = true + return { pages, signals } +} + +/** Follow one level of sitemap index, which is what most real sitemaps are. */ +async function collectFromSitemaps( + candidates: string[], + origin: string, + doFetch: typeof fetch, + timeLeft: () => number, +): Promise { + const found: string[] = [] + + for (const candidate of candidates) { + if (timeLeft() <= 0) break + const xml = await fetchText(candidate, doFetch, Math.min(LIMITS.sitemapTimeoutMs, timeLeft())) + if (!xml) continue + + const parsed = parseSitemapXml(xml) + if (!parsed.isIndex) { + found.push(...parsed.urls) + } else { + // One level down only — deep indexes are not worth the request budget. + for (const child of parsed.urls.slice(0, 3)) { + if (timeLeft() <= 0) break + const childXml = await fetchText( + child, + doFetch, + Math.min(LIMITS.sitemapTimeoutMs, timeLeft()), + ) + if (childXml) found.push(...parseSitemapXml(childXml).urls) + } + } + if (found.length) break + } + + return rankUrls(found, origin, LIMITS.maxPages) +} + +/** Same-host hrefs from a page, for the no-sitemap path. */ +function linksIn(html: string, origin: string): string[] { + const out: string[] = [] + for (const m of html.matchAll(/]*href=["']([^"']+)["']/gi)) out.push(m[1]) + return rankUrls(out, origin, LIMITS.maxPages * 2) +} + +/** + * Fetch a URL as text, or `null` if it is not usable. + * + * Enforces the timeout, the byte cap, the redirect cap, and HTML-only. Every + * redirect hop is re-validated: a public URL is allowed to redirect to + * `127.0.0.1`, and `global_fetch_strictly_public` is the backstop rather than + * the only check. + */ +async function fetchText( + url: string, + doFetch: typeof fetch, + budgetMs: number, +): Promise { + if (budgetMs <= 0) return null + + let current = url + for (let hop = 0; hop <= LIMITS.maxRedirects; hop++) { + const checked = validateUrl(current) + if (!checked.ok) return null + + // The caller decides the budget — a sitemap is worth waiting longer for + // than a page. Clamping here would silently override that. + const controller = new AbortController() + const timer = setTimeout(() => controller.abort(), budgetMs) + + try { + const res = await doFetch(checked.url, { + redirect: 'manual', + signal: controller.signal, + headers: { 'user-agent': USER_AGENT, accept: 'text/html,application/xhtml+xml,text/xml' }, + }) + + if (res.status >= 300 && res.status < 400) { + const location = res.headers.get('location') + if (!location) return null + current = new URL(location, checked.url).toString() + continue + } + + if (!res.ok) return null + + const type = res.headers.get('content-type') ?? '' + if (type && !/text\/html|xml|text\/plain/i.test(type)) return null + + // Inside the try, and before clearTimeout: the budget has to cover + // reading the body, not just receiving the headers. A server that sends + // headers instantly and then trickles bytes would otherwise hang here + // with nothing to stop it. + return await readCapped(res) + } catch { + return null + } finally { + clearTimeout(timer) + } + } + + return null +} + +/** Read a body up to the byte cap, then stop pulling. */ +async function readCapped(res: Response): Promise { + const reader = res.body?.getReader() + if (!reader) return null + + const chunks: Uint8Array[] = [] + let size = 0 + try { + for (;;) { + const { done, value } = await reader.read() + if (done) break + if (!value) continue + size += value.byteLength + chunks.push(value) + if (size >= LIMITS.maxBytes) break + } + } catch { + return null + } finally { + await reader.cancel().catch(() => {}) + } + + const buffer = new Uint8Array(size) + let offset = 0 + for (const c of chunks) { + buffer.set(c, offset) + offset += c.byteLength + } + return new TextDecoder('utf-8', { fatal: false }).decode(buffer) +} diff --git a/lib/crawl/sitemap.test.ts b/lib/crawl/sitemap.test.ts new file mode 100644 index 0000000..c8fd3c8 --- /dev/null +++ b/lib/crawl/sitemap.test.ts @@ -0,0 +1,156 @@ +import { describe, expect, it } from 'vitest' +import { + isAllowed, + parseRobots, + parseSitemapXml, + rankUrls, +} from './sitemap' + +describe('parseRobots', () => { + it('reads Sitemap directives regardless of user-agent group', () => { + const r = parseRobots(` + Sitemap: https://example.com/sitemap_index.xml + User-agent: * + Disallow: /admin + Sitemap: https://example.com/news-sitemap.xml + `) + expect(r.sitemaps).toEqual([ + 'https://example.com/sitemap_index.xml', + 'https://example.com/news-sitemap.xml', + ]) + }) + + it('only applies rules from groups that cover us', () => { + const r = parseRobots(` + User-agent: AhrefsBot + Disallow: / + + User-agent: * + Disallow: /cart + `) + // The blanket Disallow belongs to AhrefsBot, not to us. + expect(r.disallow).toEqual(['/cart']) + expect(isAllowed('/pricing', r)).toBe(true) + }) + + it('ignores comments and blank lines', () => { + const r = parseRobots(` + # a comment + User-agent: * # trailing comment + Disallow: /private + `) + expect(r.disallow).toEqual(['/private']) + }) + + it('lets an exact Allow override a Disallow', () => { + const r = parseRobots(` + User-agent: * + Disallow: /docs + Allow: /docs + `) + expect(isAllowed('/docs/intro', r)).toBe(true) + }) +}) + +describe('isAllowed', () => { + const blockAll = parseRobots('User-agent: *\nDisallow: /') + const blockSome = parseRobots('User-agent: *\nDisallow: /admin\nDisallow: /tmp*') + + it('treats Disallow: / as blocking everything', () => { + expect(isAllowed('/', blockAll)).toBe(false) + expect(isAllowed('/about', blockAll)).toBe(false) + }) + + it('blocks by prefix and trailing wildcard', () => { + expect(isAllowed('/admin/users', blockSome)).toBe(false) + expect(isAllowed('/tmpfiles/x', blockSome)).toBe(false) + expect(isAllowed('/about', blockSome)).toBe(true) + }) + + it('allows everything when there are no rules', () => { + expect(isAllowed('/anything', parseRobots(''))).toBe(true) + }) +}) + +describe('parseSitemapXml', () => { + it('detects a sitemap index — the case the spec got wrong', () => { + // WordPress/Yoast and Shopify serve this shape. Reading it as a page list + // yields zero pages, so the tool would silently fail on those sites. + const r = parseSitemapXml(` + + https://example.com/post-sitemap.xml + https://example.com/page-sitemap.xml + `) + expect(r.isIndex).toBe(true) + expect(r.urls).toHaveLength(2) + }) + + it('reads a plain urlset as pages', () => { + const r = parseSitemapXml(` + https://example.com/ + https://example.com/about + `) + expect(r.isIndex).toBe(false) + expect(r.urls).toEqual(['https://example.com/', 'https://example.com/about']) + }) + + it('handles entities and whitespace inside loc', () => { + const r = parseSitemapXml('\n https://example.com/a?x=1&y=2\n') + expect(r.urls).toEqual(['https://example.com/a?x=1&y=2']) + }) + + it('returns nothing for junk rather than throwing', () => { + expect(parseSitemapXml('404').urls).toEqual([]) + }) +}) + +describe('rankUrls', () => { + const origin = 'https://example.com' + + it('puts the homepage first and buyer-relevant pages next', () => { + const ranked = rankUrls( + [ + 'https://example.com/blog/2024/some-post', + 'https://example.com/pricing', + 'https://example.com/', + 'https://example.com/careers/engineering/backend', + ], + origin, + 10, + ) + expect(ranked[0]).toBe('https://example.com/') + expect(ranked[1]).toBe('https://example.com/pricing') + }) + + it('drops other hosts', () => { + const ranked = rankUrls(['https://example.com/a', 'https://evil.com/b'], origin, 10) + expect(ranked).toEqual(['https://example.com/a']) + }) + + it('drops non-HTML files that would waste the page budget', () => { + const ranked = rankUrls( + ['https://example.com/brochure.pdf', 'https://example.com/logo.svg', 'https://example.com/about'], + origin, + 10, + ) + expect(ranked).toEqual(['https://example.com/about']) + }) + + it('deduplicates, ignoring fragments', () => { + const ranked = rankUrls( + ['https://example.com/about', 'https://example.com/about#team', 'https://example.com/about'], + origin, + 10, + ) + expect(ranked).toHaveLength(1) + }) + + it('respects the limit', () => { + const many = Array.from({ length: 50 }, (_, i) => `https://example.com/p${i}`) + expect(rankUrls(many, origin, 20)).toHaveLength(20) + }) + + it('resolves relative URLs against the origin', () => { + expect(rankUrls(['/about'], origin, 5)).toEqual(['https://example.com/about']) + }) +}) diff --git a/lib/crawl/sitemap.ts b/lib/crawl/sitemap.ts new file mode 100644 index 0000000..e555b54 --- /dev/null +++ b/lib/crawl/sitemap.ts @@ -0,0 +1,175 @@ +/** + * Finding the pages worth reading. + * + * Everything here is pure string work — no fetching — so the awkward parts + * (sitemap indexes, robots precedence, URL ranking) are testable without a + * network. `fetch-pages.ts` does the I/O and calls into these. + * + * The spec said "fetch /sitemap.xml and take up to 20 URLs". That fails on a + * large share of real sites for two reasons this module fixes: the standard + * discovery path is the `Sitemap:` directive in robots.txt, and `sitemap.xml` + * is frequently a *sitemap index* pointing at child sitemaps rather than a list + * of pages. Parsing an index naively returns zero pages. + */ + +export type RobotsRules = { + /** Absolute URLs from `Sitemap:` directives. Order preserved. */ + sitemaps: string[] + /** Disallowed path prefixes that apply to us. */ + disallow: string[] +} + +/** + * Parse robots.txt for the directives we honour. + * + * ponytail: implements the common subset, not RFC 9309 — `User-agent` grouping, + * `Disallow`, `Allow` (as a disallow override), and `Sitemap`. No wildcard + * expansion beyond a trailing `*`, no crawl-delay. That covers what real sites + * use to block crawlers. If we ever see a site whose rules we misread, the + * upgrade path is a real robots parser, not more regex here. + */ +export function parseRobots(text: string): RobotsRules { + const sitemaps: string[] = [] + const disallow: string[] = [] + const allow: string[] = [] + + // Which user-agent group we are currently inside. We obey `*` and any group + // naming us; everything else is skipped. + let applies = false + + for (const rawLine of text.split(/\r?\n/)) { + const line = rawLine.split('#')[0].trim() + if (!line) continue + + const sep = line.indexOf(':') + if (sep === -1) continue + + const field = line.slice(0, sep).trim().toLowerCase() + const value = line.slice(sep + 1).trim() + if (!value) continue + + // Sitemap is a global directive — it is not scoped to a user-agent group. + if (field === 'sitemap') { + sitemaps.push(value) + continue + } + + if (field === 'user-agent') { + const ua = value.toLowerCase() + applies = ua === '*' || ua.includes('epyc') + continue + } + + if (!applies) continue + if (field === 'disallow') disallow.push(value) + if (field === 'allow') allow.push(value) + } + + // An `Allow` that exactly matches a `Disallow` wins, per the usual precedence. + const overridden = new Set(allow) + return { + sitemaps, + disallow: disallow.filter((d) => !overridden.has(d)), + } +} + +/** Would robots.txt let us fetch this path? `Disallow: /` blocks everything. */ +export function isAllowed(pathname: string, rules: RobotsRules): boolean { + return !rules.disallow.some((rule) => { + if (rule === '/') return true + // Trailing `*` is the only wildcard worth supporting in practice. + const prefix = rule.endsWith('*') ? rule.slice(0, -1) : rule + return prefix !== '' && pathname.startsWith(prefix) + }) +} + +export type SitemapParse = { + /** `` values found. */ + urls: string[] + /** True when this was a `` — the URLs are child sitemaps. */ + isIndex: boolean +} + +/** + * Pull `` values out of a sitemap or sitemap index. + * + * ponytail: regex, not an XML parser. Sitemaps are a fixed, shallow schema and + * we want exactly one element from them. Workers has no DOMParser, and adding + * an XML dependency to read one tag would be the definition of over-building. + */ +export function parseSitemapXml(xml: string): SitemapParse { + const isIndex = /]/i.test(xml) + const urls: string[] = [] + + for (const match of xml.matchAll(/\s*([\s\S]*?)\s*<\/loc>/gi)) { + const value = decodeXmlEntities(match[1].trim()) + if (value) urls.push(value) + } + + return { urls, isIndex } +} + +function decodeXmlEntities(s: string): string { + return s + .replace(/</g, '<') + .replace(/>/g, '>') + .replace(/"/g, '"') + .replace(/'/g, "'") + .replace(/&/g, '&') +} + +/** Slugs a buyer looks for first — these are also what the report scores. */ +const PRIORITY = /(about|service|product|pricing|price|plans|contact|work|case|solution|team)/i + +/** + * Choose which pages to read, best first. + * + * Same-host only, shallow paths before deep ones, buyer-relevant slugs before + * everything else, the homepage always first. Non-HTML extensions are dropped — + * a PDF or an image is a wasted fetch out of a budget of 20. + */ +export function rankUrls(urls: string[], origin: string, limit: number): string[] { + let host: string + try { + host = new URL(origin).host + } catch { + return [] + } + + const seen = new Set() + const candidates: { url: string; score: number }[] = [] + + for (const raw of urls) { + let u: URL + try { + u = new URL(raw, origin) + } catch { + continue + } + + if (u.host !== host) continue + if (u.protocol !== 'http:' && u.protocol !== 'https:') continue + if (/\.(pdf|jpe?g|png|gif|svg|webp|avif|zip|mp4|mp3|css|js|xml|json)$/i.test(u.pathname)) continue + + u.hash = '' + const key = u.toString() + if (seen.has(key)) continue + seen.add(key) + + const depth = u.pathname.split('/').filter(Boolean).length + // Lower is better: homepage 0, priority slugs beat depth, deep pages last. + const score = (u.pathname === '/' ? -100 : 0) + (PRIORITY.test(u.pathname) ? -10 : 0) + depth + + candidates.push({ url: key, score }) + } + + return candidates + .sort((a, b) => a.score - b.score) + .slice(0, limit) + .map((c) => c.url) +} + +/** Where to look for a sitemap when robots.txt names none. */ +export function fallbackSitemapUrls(origin: string): string[] { + return [new URL('/sitemap.xml', origin).toString(), new URL('/sitemap_index.xml', origin).toString()] +} diff --git a/lib/crawl/validate-url.test.ts b/lib/crawl/validate-url.test.ts new file mode 100644 index 0000000..bcdebf3 --- /dev/null +++ b/lib/crawl/validate-url.test.ts @@ -0,0 +1,96 @@ +import { describe, expect, it } from 'vitest' +import { validateUrl } from './validate-url' + +/** + * The other place a silent bug is expensive: this function is what stands + * between an anonymous visitor's text box and a `fetch` from our Worker. + */ + +describe('validateUrl — accepts', () => { + it('a bare domain, adding https', () => { + const r = validateUrl('example.com') + expect(r.ok).toBe(true) + if (r.ok) { + expect(r.url).toBe('https://example.com/') + expect(r.host).toBe('example.com') + } + }) + + it('a full URL with a path', () => { + const r = validateUrl('https://example.com/pricing') + expect(r.ok).toBe(true) + if (r.ok) expect(r.url).toBe('https://example.com/pricing') + }) + + it('http as well as https', () => { + expect(validateUrl('http://example.com').ok).toBe(true) + }) + + it('subdomains and surrounding whitespace', () => { + const r = validateUrl(' https://www.example.co.uk/about ') + expect(r.ok).toBe(true) + if (r.ok) expect(r.host).toBe('www.example.co.uk') + }) + + it('an explicit standard port', () => { + expect(validateUrl('https://example.com:443/').ok).toBe(true) + }) + + it('but strips the fragment, so the reuse cache does not split', () => { + const r = validateUrl('https://example.com/about#team') + expect(r.ok).toBe(true) + if (r.ok) expect(r.url).toBe('https://example.com/about') + }) +}) + +describe('validateUrl — rejects', () => { + it('empty input', () => { + expect(validateUrl(' ').ok).toBe(false) + }) + + it('IPv4 literals, including cloud metadata', () => { + expect(validateUrl('169.254.169.254').ok).toBe(false) + expect(validateUrl('http://127.0.0.1/').ok).toBe(false) + expect(validateUrl('https://10.0.0.5/admin').ok).toBe(false) + }) + + it('IPv6 literals', () => { + expect(validateUrl('http://[::1]/').ok).toBe(false) + expect(validateUrl('http://[fe80::1]/').ok).toBe(false) + }) + + it('localhost and internal suffixes', () => { + expect(validateUrl('localhost').ok).toBe(false) + expect(validateUrl('http://localhost:3000').ok).toBe(false) + expect(validateUrl('printer.local').ok).toBe(false) + expect(validateUrl('vault.internal').ok).toBe(false) + }) + + it('single-label hosts that only resolve inside a network', () => { + expect(validateUrl('intranet').ok).toBe(false) + expect(validateUrl('http://wiki/').ok).toBe(false) + }) + + it('non-web schemes', () => { + expect(validateUrl('file:///etc/passwd').ok).toBe(false) + expect(validateUrl('ftp://example.com').ok).toBe(false) + expect(validateUrl('javascript:alert(1)').ok).toBe(false) + expect(validateUrl('data:text/html,

hi').ok).toBe(false) + }) + + it('credentials in the authority, which disguise the real host', () => { + expect(validateUrl('https://example.com@169.254.169.254/').ok).toBe(false) + expect(validateUrl('https://user:pass@example.com/').ok).toBe(false) + }) + + it('non-standard ports that would reach internal services', () => { + expect(validateUrl('https://example.com:8080/').ok).toBe(false) + expect(validateUrl('http://example.com:22/').ok).toBe(false) + }) + + it('with a reason a visitor can act on, never a raw error', () => { + const r = validateUrl('127.0.0.1') + expect(r.ok).toBe(false) + if (!r.ok) expect(r.reason).toMatch(/domain name/i) + }) +}) diff --git a/lib/crawl/validate-url.ts b/lib/crawl/validate-url.ts new file mode 100644 index 0000000..08e7c18 --- /dev/null +++ b/lib/crawl/validate-url.ts @@ -0,0 +1,78 @@ +/** + * Safety check for a visitor-supplied URL, before the crawler fetches it. + * + * This is a *syntactic* gate. It cannot resolve DNS — Workers has no resolver + * API — so a hostname that resolves to a private address still passes here. + * That case is covered at fetch time by the `global_fetch_strictly_public` + * compatibility flag in wrangler.jsonc, which blocks Workers `fetch` to private + * and internal addresses. The two together are the SSRF mitigation; neither is + * sufficient alone, so do not remove either believing the other covers it. + * + * See docs/ai-chatbot-architecture.md §2.2. + */ + +export type UrlCheck = + | { ok: true; url: string; host: string } + | { ok: false; reason: string } + +/** Dotted-quad, with or without a port. Catches 127.0.0.1, 169.254.169.254, etc. */ +const IPV4 = /^\d{1,3}(\.\d{1,3}){3}$/ + +/** Suffixes that only ever name something inside a network. */ +const INTERNAL_SUFFIXES = ['.local', '.localhost', '.internal', '.home.arpa', '.onion'] + +/** A crawl seed is a public web page, so only the web's own ports. */ +const ALLOWED_PORTS = new Set(['', '80', '443']) + +export function validateUrl(input: string): UrlCheck { + const trimmed = input.trim() + if (!trimmed) return { ok: false, reason: 'Enter a website address.' } + + // Visitors type "example.com", not "https://example.com". + const withScheme = /^[a-z][a-z0-9+.-]*:\/\//i.test(trimmed) ? trimmed : `https://${trimmed}` + + let url: URL + try { + url = new URL(withScheme) + } catch { + return { ok: false, reason: "That doesn't look like a website address." } + } + + if (url.protocol !== 'http:' && url.protocol !== 'https:') { + return { ok: false, reason: 'Only http and https addresses can be read.' } + } + + // user:pass@host — a classic way to disguise the real host from a reader. + if (url.username || url.password) { + return { ok: false, reason: 'Addresses with login details are not accepted.' } + } + + if (!ALLOWED_PORTS.has(url.port)) { + return { ok: false, reason: 'Only standard web ports can be read.' } + } + + const host = url.hostname.toLowerCase() + + // URL wraps IPv6 literals in brackets — [::1], [fe80::1]. + if (host.startsWith('[')) { + return { ok: false, reason: 'Enter a domain name, not an IP address.' } + } + + if (IPV4.test(host)) { + return { ok: false, reason: 'Enter a domain name, not an IP address.' } + } + + if (host === 'localhost' || INTERNAL_SUFFIXES.some((s) => host.endsWith(s))) { + return { ok: false, reason: 'That address is not reachable from the public internet.' } + } + + // Single-label hosts ("intranet", "wiki") only resolve inside a network. + if (!host.includes('.')) { + return { ok: false, reason: 'Enter a full domain, like example.com.' } + } + + // Fragments are meaningless to a crawler and would split the reuse cache. + url.hash = '' + + return { ok: true, url: url.toString(), host } +} diff --git a/lib/image-loader.ts b/lib/image-loader.ts index ea61379..6c4f1f3 100644 --- a/lib/image-loader.ts +++ b/lib/image-loader.ts @@ -30,9 +30,18 @@ export default function imageLoader({ src, width, quality }: LoaderArgs): string // `next dev` isn't behind a Cloudflare zone — cdn-cgi doesn't work. // Bare Strapi paths need the media base prepended so they're fetchable. + // + // The `?w=` is carried purely so Next can see the width in the returned URL. + // Without it every logs "loader property that does not implement + // width" on every dev page load. Nothing serves a different file for it — + // static assets and the media host both ignore the unknown parameter — and + // this branch never runs in production. if (process.env.NODE_ENV === 'development') { - if (!src.startsWith('/images/') && !src.startsWith('http')) return `${mediaBase}${src}` - return src + const devWidth = `${src.includes('?') ? '&' : '?'}w=${width}` + if (!src.startsWith('/images/') && !src.startsWith('http')) { + return `${mediaBase}${src}${devWidth}` + } + return `${src}${devWidth}` } // Public folder assets (/images/site/...) — cdn-cgi on the main zone. diff --git a/lib/tools/chatbot/diagnosis.ts b/lib/tools/chatbot/diagnosis.ts new file mode 100644 index 0000000..72dc82e --- /dev/null +++ b/lib/tools/chatbot/diagnosis.ts @@ -0,0 +1,166 @@ +/** + * The AI chatbot demo's report. + * + * Five dimensions. Three come from the shared site checks and cost nothing; + * Answerability and Coverage need judgement and share a single model call. If + * that call fails we still render the other three — a partial report converts, + * an error state does not. + * + * Every number points at something. No score is emitted without the evidence + * underneath it, because the whole pitch is "here is what your own site says". + * + * Everything tool-agnostic lives in ../site-checks.ts and is shared with the + * Website Grader and the llms.txt Generator. Only the chatbot-specific half — + * the ten buyer questions and the page types — is here. + */ + +import { buyerQuestions } from '@/data/buyer-questions' +import { SCORING_CHAIN, completeJson } from '../models' +import type { StoredPage } from '../session' +import { runSiteChecks, type CrawlSignals, type SiteChecks, type SitePage } from '../site-checks' + +export type { CrawlSignals, SitePage } + +export type Diagnosis = SiteChecks & { + /** Null until the model call lands. */ + answerability: { + answered: number + total: number + unanswered: { id: string; question: string }[] + } | null + coverage: { present: string[]; missing: string[] } | null + /** True while answerability and coverage are missing. */ + partial: boolean + scoredAt: string +} + +/** The measured three, with the judged two left null for the background call. */ +export function scoreDeterministic( + pages: SitePage[], + signals: CrawlSignals, + signalsKnown = true, +): Diagnosis { + return { + ...runSiteChecks(pages, signals, signalsKnown), + answerability: null, + coverage: null, + partial: true, + scoredAt: new Date().toISOString(), + } +} + +/** + * Fold the judged half into a stored diagnosis, guaranteeing a complete one. + * + * The naive `{...existing, ...judged}` silently produces a report with an + * answerability score and no structure or crawlability when `existing` is + * missing or malformed — which the UI then dereferences and dies on. Rebuild + * the measured half from the pages rather than trusting what was stored. + */ +export function mergeJudged( + existing: Partial | null, + judged: Pick, + pages: SitePage[], + signalsKnown = false, +): Diagnosis { + const base = + existing?.structure && existing.crawlability && existing.specificity + ? (existing as Diagnosis) + : scoreDeterministic(pages, {}, signalsKnown) + + return { ...base, ...judged, partial: false } +} + +const PAGE_TYPES = [ + 'what they do', + 'who they serve', + 'pricing or process', + 'proof — results, case studies or named clients', + 'contact', +] as const + +type ModelVerdict = { + questions: { id: string; answerable: boolean; evidence: string }[] + pageTypesPresent: string[] +} + +/** + * Answerability and Coverage — the two that need judgement — in one call. + * + * Asking for evidence per question is not decoration: it forces the model to + * point at the text before it claims a question is answered, which is what + * stops a confident "yes" on a page that says nothing. It also makes a wrong + * score debuggable rather than mysterious. + */ +export async function scoreWithModel( + apiKey: string, + host: string, + pages: StoredPage[], + opts: { allowPaid?: boolean } = {}, +): Promise> { + const corpus = pages + .map((p) => `## ${p.title || 'Untitled'}\nURL: ${p.url}\n\n${p.text}`) + .join('\n\n---\n\n') + .slice(0, 300_000) + + const questionList = buyerQuestions + .map((q) => `- id "${q.id}": ${q.question} (answered if the site shows: ${q.looksLike})`) + .join('\n') + + const result = await completeJson({ + apiKey, + chain: SCORING_CHAIN, + allowPaid: opts.allowPaid, + maxTokens: 3000, + messages: [ + { + role: 'system', + content: `You audit whether a company's website answers the questions a buyer asks before getting in touch. You judge ONLY from the supplied page text. You never use outside knowledge about the company. You are strict but fair: a question counts as answered when a buyer could act on what the site actually says, even if it is brief. It does not count when the site only gestures at the topic without specifics. Reply with JSON only.`, + }, + { + role: 'user', + content: `Website: ${host} + +For each question below, decide whether the site text answers it. Quote the exact words that answer it, or an empty string if nothing does. + +Questions: +${questionList} + +Also list which of these page types the site clearly has: ${PAGE_TYPES.map((t) => `"${t}"`).join(', ')}. + +Reply with exactly this JSON shape and nothing else: +{"questions":[{"id":"","answerable":true|false,"evidence":""}],"pageTypesPresent":[""]} + +--- WEBSITE CONTENT --- +${corpus}`, + }, + ], + }) + + const byId = new Map(result.questions?.map((q) => [q.id, q]) ?? []) + + // Trust the model's verdict only where it produced evidence. A claimed + // "answerable" with nothing quoted is the failure mode this guards against. + const unanswered = buyerQuestions + .filter((q) => { + const verdict = byId.get(q.id) + return !verdict?.answerable || !verdict.evidence?.trim() + }) + .map((q) => ({ id: q.id, question: q.question })) + + const present = (result.pageTypesPresent ?? []).filter((t) => + (PAGE_TYPES as readonly string[]).includes(t), + ) + + return { + answerability: { + answered: buyerQuestions.length - unanswered.length, + total: buyerQuestions.length, + unanswered, + }, + coverage: { + present, + missing: PAGE_TYPES.filter((t) => !present.includes(t)), + }, + } +} diff --git a/lib/tools/chatbot/embed.ts b/lib/tools/chatbot/embed.ts new file mode 100644 index 0000000..e9fb57d --- /dev/null +++ b/lib/tools/chatbot/embed.ts @@ -0,0 +1,136 @@ +/** + * Embed keys: minting, lookup, and the Origin check that secures them. + * + * The key is public — it sits in the HTML of the customer's site, so anyone + * can read it. Security is binding plus a cap, not concealment: + * + * 1. A key minted for acme.com only answers requests whose Origin is + * acme.com or www.acme.com. CORS is set to that host, never `*`. + * 2. Each key carries its own daily message counter. Origin can be forged by + * a non-browser client, so the cap is what bounds the worst case — a + * bounded amount of free inference, not an open tap. + * 3. Keys can be revoked with one row update. + */ + +import { bumpCounter } from '../counters' + +export type EmbedRow = { + key: string + session_id: string + bound_host: string + status: 'active' | 'revoked' + attribution: number + crawled_at: string +} + +/** Messages per embed per day. A runaway backstop, not a product limit. */ +export const EMBED_DAILY_MESSAGES = 50 + +const KEY_PREFIX = 'ek_live_' + +/** 128 bits of randomness, hex encoded. Not a secret, but must be unguessable. */ +export function mintKey(): string { + const bytes = crypto.getRandomValues(new Uint8Array(16)) + return KEY_PREFIX + [...bytes].map((b) => b.toString(16).padStart(2, '0')).join('') +} + +export function isEmbedKey(value: string): boolean { + return /^ek_live_[0-9a-f]{32}$/.test(value) +} + +/** + * Does this Origin belong to the host the key was minted for? + * + * Apex and `www` only. Subdomains are deliberately not accepted: a key issued + * for acme.com should not answer for anything.acme.com, because we never + * crawled those and the bot would confidently answer from the wrong corpus. + */ +export function originAllowed( + origin: string | null, + boundHost: string, + opts: { allowAny?: boolean } = {}, +): boolean { + if (!origin) return false + + // Local development only. A key is bound to the crawled site, so a widget + // can never be previewed from localhost without this. Gated on an env var + // that is unset in staging and production — see cloudflare-env.secrets.d.ts. + if (opts.allowAny) return true + + let host: string + try { + const url = new URL(origin) + if (url.protocol !== 'https:' && url.protocol !== 'http:') return false + host = url.host.toLowerCase() + } catch { + return false + } + + const apex = boundHost.toLowerCase().replace(/^www\./, '') + return host === apex || host === `www.${apex}` +} + +/** CORS headers for a bound embed. Never `*` — the allowlist is one host. */ +export function corsHeaders(origin: string): Record { + return { + 'access-control-allow-origin': origin, + 'access-control-allow-methods': 'POST, OPTIONS', + 'access-control-allow-headers': 'content-type', + 'access-control-max-age': '86400', + vary: 'Origin', + } +} + +export async function createEmbed( + db: D1Database, + input: { key: string; sessionId: string; boundHost: string; email: string | null; crawledAt: string }, +): Promise { + await db + .prepare( + `INSERT INTO tool_embeds (key, session_id, bound_host, email, crawled_at) + VALUES (?, ?, ?, ?, ?)`, + ) + .bind(input.key, input.sessionId, input.boundHost, input.email, input.crawledAt) + .run() +} + +/** An existing embed for this session, so re-claiming returns the same key. */ +export async function findEmbedBySession( + db: D1Database, + sessionId: string, +): Promise { + return db + .prepare( + `SELECT key, session_id, bound_host, status, attribution, crawled_at + FROM tool_embeds WHERE session_id = ?`, + ) + .bind(sessionId) + .first() +} + +export async function findEmbedByKey(db: D1Database, key: string): Promise { + return db + .prepare( + `SELECT key, session_id, bound_host, status, attribution, crawled_at + FROM tool_embeds WHERE key = ?`, + ) + .bind(key) + .first() +} + +/** Consume one of this key's daily messages. False means the cap is reached. */ +export async function consumeEmbedMessage(db: D1Database, key: string): Promise { + return bumpCounter(db, `embed:${key}`, EMBED_DAILY_MESSAGES) +} + +export async function touchEmbed(db: D1Database, key: string): Promise { + await db + .prepare('UPDATE tool_embeds SET last_message_at = CURRENT_TIMESTAMP WHERE key = ?') + .bind(key) + .run() +} + +/** The snippet a customer pastes. Kept in one place so the docs cannot drift. */ +export function embedSnippet(key: string, origin: string): string { + return `` +} diff --git a/lib/tools/chatbot/prompt.ts b/lib/tools/chatbot/prompt.ts new file mode 100644 index 0000000..5fc9ca2 --- /dev/null +++ b/lib/tools/chatbot/prompt.ts @@ -0,0 +1,85 @@ +/** + * The system prompt and corpus assembly. + * + * Rule 2 below is the load-bearing one. The honest "I could not find that on + * the site" answers are the raw material the report is built from, and they + * are the sales argument. A bot that bluffs destroys the whole mechanic — it + * makes a thin site look adequate, which is the opposite of the point. + */ + +import type { ChatMessage } from '../models' +import type { StoredPage } from '../session' + +/** Rough budget for the corpus. Lightning's window is 1M, so this is slack. */ +const MAX_CORPUS_CHARS = 400_000 + +export function systemPrompt(host: string, corpus: string): string { + return `You are the assistant on ${host}'s website. You help visitors of that site, answering from the site's own pages, which are supplied below. + +WHO YOU ARE + +You are ${host}'s website assistant and nothing else. You are not a general-purpose assistant, and you never present yourself as one. + +- If asked who or what you are, say you are the assistant for ${host} and that you can answer questions about the company and what it offers. Nothing more. +- NEVER say you are an AI, a language model, a chatbot, or a bot. NEVER name a model, a vendor, or the technology behind you — not Nemotron, not NVIDIA, not any other. If pushed, say only that you are ${host}'s website assistant. +- NEVER mention EPYC. NEVER say you are a demo, a trial, a diagnostic, or an evaluation. +- If asked what you can do, answer in terms of THIS website — the things the pages below actually cover — not a list of general abilities like translation, coding, or creative writing. You do none of those things. + + Wrong: "I can answer questions about the company from the site's pages, all grounded in the website content provided. I don't have general-purpose abilities like translation or coding, and I only work from the supplied text." + Right: "I can tell you about our products, pricing and support. What would you like to know?" + + The wrong answer describes how you work and lists what you cannot do. The right one names what this company offers and stops. +- Never critique this website, its structure, or its content, even if asked to. + +WHAT YOU ANSWER + +1. Answer only from the supplied page text. Never use outside knowledge about this company, its industry, or its competitors. If you happen to recognise the company, ignore what you know. + +2. When the supplied text does not contain the answer, say so plainly and name what is missing. For example: "I couldn't find that on the site — there's no pricing page, and the services pages don't mention cost." Do not guess, do not infer, do not pad with generalities, and never suggest what the answer is "likely" to be. Point the visitor at whatever contact route the site provides, if there is one. + +3. Only answer questions about ${host}, what it does, and what it offers. If asked about anything unrelated — general knowledge, other companies, writing or coding help — say that you can only help with questions about ${host}, and offer something you can actually answer from its pages. + +4. On a greeting, greet back in one short line and say what you can help with, grounded in what this site is actually about. Do not ask an open-ended "how can I help you today?" with no context. + +HOW YOU WRITE + +Short — two or three sentences unless asked for detail. Plain language, no marketing tone, no bullet lists unless the question genuinely calls for one. Write as part of ${host}, using "we" for the company where it reads naturally. + +Never describe your own workings. The visitor cannot see anything that was given to you, so never refer to "the supplied text", "the pages above", "the website content provided", "the corpus", or "what I was given". Say "on our site", "on our pricing page", or "I couldn't find that on our site" instead. A visitor should only ever hear about the website, never about how you read it. + +--- WEBSITE CONTENT BEGINS --- +${corpus} +--- WEBSITE CONTENT ENDS ---` +} + +/** One block per page: title, URL, then its text. */ +export function buildCorpus(pages: StoredPage[]): string { + const blocks: string[] = [] + let total = 0 + + for (const page of pages) { + if (!page.text?.trim()) continue + const block = `## ${page.title || 'Untitled'}\nURL: ${page.url}\n\n${page.text}` + if (total + block.length > MAX_CORPUS_CHARS) break + blocks.push(block) + total += block.length + } + + return blocks.join('\n\n---\n\n') +} + +export type Turn = { role: 'user' | 'assistant'; content: string } + +/** System prompt + prior turns + the new question. */ +export function buildMessages( + host: string, + pages: StoredPage[], + history: Turn[], + question: string, +): ChatMessage[] { + return [ + { role: 'system', content: systemPrompt(host, buildCorpus(pages)) }, + ...history.map((t) => ({ role: t.role, content: t.content }) as ChatMessage), + { role: 'user', content: question }, + ] +} diff --git a/lib/tools/chatbot/schema.ts b/lib/tools/chatbot/schema.ts new file mode 100644 index 0000000..b45f5fa --- /dev/null +++ b/lib/tools/chatbot/schema.ts @@ -0,0 +1,56 @@ +import { z } from 'zod' + +/** + * Request schemas for the AI chatbot tool's routes. + * + * Kept beside the tool rather than in the route files, matching + * lib/contact/schema.ts and lib/workshop/schema.ts — the routes parse, the + * schema lives here. + */ + +export const crawlSchema = z.object({ + url: z.string().min(1).max(2048), + /** Set by the "read my site again" button — bypasses the 24h reuse cache. */ + force: z.boolean().optional().default(false), +}) + +export const messageSchema = z.object({ + sessionId: z.string().uuid(), + message: z.string().min(1).max(2000), +}) + +/** Claiming an embed. Email is captured, not yet verified — see the route. */ +export const claimSchema = z.object({ + sessionId: z.string().uuid(), + email: z.string().email().max(320), +}) + +/** A message sent to a live widget on a customer's own site. */ +export const embedMessageSchema = z.object({ + key: z.string().max(64), + message: z.string().min(1).max(2000), + /** Prior turns, held by the widget rather than the server. */ + history: z + .array(z.object({ role: z.enum(['user', 'assistant']), content: z.string().max(4000) })) + .max(20) + .optional() + .default([]), +}) + +/** Asking for a verification code. */ +export const verifySendSchema = z.object({ + sessionId: z.string().uuid(), + email: z.string().email().max(320), +}) + +/** Submitting one. */ +export const verifyCheckSchema = z.object({ + sessionId: z.string().uuid(), + email: z.string().email().max(320), + code: z.string().regex(/^\d{6}$/, 'Enter the 6-digit code.'), +}) + +export type CrawlInput = z.infer +export type MessageInput = z.infer +export type ClaimInput = z.infer +export type EmbedMessageInput = z.infer diff --git a/lib/tools/chatbot/verification.ts b/lib/tools/chatbot/verification.ts new file mode 100644 index 0000000..92c0133 --- /dev/null +++ b/lib/tools/chatbot/verification.ts @@ -0,0 +1,188 @@ +/** + * Email verification for claiming an embed. + * + * A six-digit code is only a million possibilities, so the controls around it + * carry the security, not the code itself: + * + * - stored as an HMAC, so a database read yields nothing usable + * - 10 minute expiry + * - 5 attempts, then the code is dead + * - single use + * - 3 codes per email per day, 3 per session + * + * The daily caps matter for a second reason: an endpoint that emails an + * arbitrary address on request is a spam cannon. Without caps, someone can + * mail-bomb a person through us, or burn our sending quota and get the domain + * blocked. + */ + +import { bumpCounter, underLimit } from '../counters' + +export const CODE_TTL_MINUTES = 10 +export const MAX_ATTEMPTS = 5 +export const CODES_PER_EMAIL_PER_DAY = 3 +export const CODES_PER_SESSION = 3 + +export type VerificationRow = { + id: string + session_id: string + email: string + code_hash: string + expires_at: string + attempts: number + consumed_at: string | null +} + +/** Six digits, uniformly distributed. `Math.random()` is not acceptable here. */ +export function generateCode(): string { + const buf = crypto.getRandomValues(new Uint32Array(1)) + return String(buf[0] % 1_000_000).padStart(6, '0') +} + +/** + * HMAC the code before storing it. + * + * A plain hash of six digits is a lookup table of a million entries — anyone + * with database access could reverse every live code instantly. The pepper is + * a Worker secret, so the stored value is only checkable by us. + */ +export async function hashCode(code: string, pepper: string): Promise { + const enc = new TextEncoder() + const key = await crypto.subtle.importKey( + 'raw', + enc.encode(pepper), + { name: 'HMAC', hash: 'SHA-256' }, + false, + ['sign'], + ) + const sig = await crypto.subtle.sign('HMAC', key, enc.encode(code)) + return [...new Uint8Array(sig)].map((b) => b.toString(16).padStart(2, '0')).join('') +} + +/** Normalised so casing and stray spaces cannot dodge the per-email cap. */ +export function normaliseEmail(email: string): string { + return email.trim().toLowerCase() +} + +export type IssueResult = + | { ok: true; code: string; expiresAt: string } + | { ok: false; reason: 'email-capped' | 'session-capped' } + +/** + * Issue a code, subject to both daily caps. + * + * Caps are consumed through `tool_counters` — one atomic statement each, the + * same mechanism the crawl and message limits use — rather than by counting + * rows, which would race. + */ +export async function issueCode( + db: D1Database, + input: { sessionId: string; email: string; pepper: string }, +): Promise { + const email = normaliseEmail(input.email) + const emailKey = `verify-email:${await hashCode(email, input.pepper)}` + const sessionKey = `verify-session:${input.sessionId}` + + // Check both before consuming either. Bumping the email counter first meant + // a request rejected by the session cap still burned one of that address's + // three daily slots — the caller got an error and paid for it anyway. + // + // Check-then-consume is not atomic; under a race the worst case is one extra + // code, which is the right way to be wrong for a limit whose purpose is + // stopping bulk abuse rather than counting exactly. + if (!(await underLimit(db, emailKey, CODES_PER_EMAIL_PER_DAY))) { + return { ok: false, reason: 'email-capped' } + } + if (!(await underLimit(db, sessionKey, CODES_PER_SESSION))) { + return { ok: false, reason: 'session-capped' } + } + + await bumpCounter(db, emailKey, CODES_PER_EMAIL_PER_DAY) + await bumpCounter(db, sessionKey, CODES_PER_SESSION) + + const code = generateCode() + const expiresAt = new Date(Date.now() + CODE_TTL_MINUTES * 60_000).toISOString() + + await db + .prepare( + `INSERT INTO tool_verifications (id, session_id, email, code_hash, expires_at) + VALUES (?, ?, ?, ?, ?)`, + ) + .bind(crypto.randomUUID(), input.sessionId, email, await hashCode(code, input.pepper), expiresAt) + .run() + + return { ok: true, code, expiresAt } +} + +export type CheckResult = + | { ok: true } + | { ok: false; reason: 'no-code' | 'expired' | 'too-many-attempts' | 'wrong-code' } + +/** + * Check a submitted code and consume it on success. + * + * The attempt counter is incremented in one conditional statement, so parallel + * guesses cannot slip past the limit — the same shape as the daily counters. + */ +export async function checkCode( + db: D1Database, + input: { sessionId: string; email: string; code: string; pepper: string }, +): Promise { + const email = normaliseEmail(input.email) + + const row = await db + .prepare( + `SELECT id, session_id, email, code_hash, expires_at, attempts, consumed_at + FROM tool_verifications + WHERE session_id = ? AND email = ? AND consumed_at IS NULL + ORDER BY created_at DESC LIMIT 1`, + ) + .bind(input.sessionId, email) + .first() + + if (!row) return { ok: false, reason: 'no-code' } + if (new Date(row.expires_at).getTime() < Date.now()) return { ok: false, reason: 'expired' } + + // Spend an attempt first, atomically. A wrong guess must cost something even + // if everything after this throws. + const spend = await db + .prepare( + `UPDATE tool_verifications SET attempts = attempts + 1 + WHERE id = ? AND attempts < ? AND consumed_at IS NULL`, + ) + .bind(row.id, MAX_ATTEMPTS) + .run() + + if ((spend.meta.changes ?? 0) === 0) return { ok: false, reason: 'too-many-attempts' } + + const submitted = await hashCode(input.code.trim(), input.pepper) + if (submitted !== row.code_hash) return { ok: false, reason: 'wrong-code' } + + // Single use: the same statement that marks it consumed is the one that + // proves it had not been consumed already. + const consume = await db + .prepare('UPDATE tool_verifications SET consumed_at = CURRENT_TIMESTAMP WHERE id = ? AND consumed_at IS NULL') + .bind(row.id) + .run() + + if ((consume.meta.changes ?? 0) === 0) return { ok: false, reason: 'no-code' } + + return { ok: true } +} + +/** Has this session verified this address? Gates the embed mint. */ +export async function isVerified( + db: D1Database, + sessionId: string, + email: string, +): Promise { + const row = await db + .prepare( + `SELECT id FROM tool_verifications + WHERE session_id = ? AND email = ? AND consumed_at IS NOT NULL LIMIT 1`, + ) + .bind(sessionId, normaliseEmail(email)) + .first<{ id: string }>() + + return Boolean(row) +} diff --git a/lib/tools/counters.test.ts b/lib/tools/counters.test.ts new file mode 100644 index 0000000..6b330f4 --- /dev/null +++ b/lib/tools/counters.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it } from 'vitest' +import { DatabaseSync } from 'node:sqlite' +import { bumpCounter, underLimit, utcDay } from './counters' + +/** + * The counter is the one piece of this tool where a silent bug is expensive: + * too permissive and someone points a script at us, too strict and the tool + * tells every visitor it is full. So it gets a real test. + * + * ponytail: tested against node:sqlite rather than @cloudflare/vitest-pool-workers. + * D1 *is* SQLite, and what is at risk here is the semantics of one conditional + * upsert — `changes` on a cold day versus at the cap — which is engine + * behaviour, not binding behaviour. Costs one dev dependency instead of a test + * harness. Move to the workers pool if we ever need to test D1-specific + * behaviour like batch() or session bookmarks. + */ + +/** Minimal stand-in for the slice of D1Database the counter functions call. */ +function fakeD1() { + const db = new DatabaseSync(':memory:') + db.exec(` + CREATE TABLE tool_counters ( + day TEXT NOT NULL, + key TEXT NOT NULL, + n INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (day, key) + ); + `) + + return { + prepare(sql: string) { + const stmt = db.prepare(sql) + return { + bind(...values: unknown[]) { + return { + async run() { + const r = stmt.run(...(values as never[])) + return { meta: { changes: Number(r.changes) } } + }, + async first() { + return (stmt.get(...(values as never[])) ?? null) as T | null + }, + } + }, + } + }, + } as unknown as D1Database +} + +const DAY = '2026-08-14' + +describe('utcDay', () => { + it('formats as YYYY-MM-DD in UTC', () => { + expect(utcDay(new Date('2026-08-14T23:30:00Z'))).toBe('2026-08-14') + }) + + it('rolls over on the UTC boundary, not the local one', () => { + expect(utcDay(new Date('2026-08-15T00:00:01Z'))).toBe('2026-08-15') + }) +}) + +describe('bumpCounter', () => { + it('allows the first call of a new day, when no row exists yet', async () => { + // The bug this guards: `UPDATE ... WHERE n < ?` changes zero rows here, + // which reads as "capped" — so the tool would report full every midnight. + const db = fakeD1() + await expect(bumpCounter(db, 'global-messages', 200, DAY)).resolves.toBe(true) + }) + + it('allows exactly `limit` calls, then refuses', async () => { + const db = fakeD1() + const results: boolean[] = [] + for (let i = 0; i < 5; i++) { + results.push(await bumpCounter(db, 'ip:abc', 3, DAY)) + } + expect(results).toEqual([true, true, true, false, false]) + }) + + it('stays refused once capped — no drift back under the limit', async () => { + const db = fakeD1() + for (let i = 0; i < 10; i++) await bumpCounter(db, 'ip:abc', 2, DAY) + await expect(bumpCounter(db, 'ip:abc', 2, DAY)).resolves.toBe(false) + expect(await underLimit(db, 'ip:abc', 2, DAY)).toBe(false) + }) + + it('counts each key separately', async () => { + const db = fakeD1() + await bumpCounter(db, 'ip:aaa', 1, DAY) + expect(await bumpCounter(db, 'ip:aaa', 1, DAY)).toBe(false) + expect(await bumpCounter(db, 'ip:bbb', 1, DAY)).toBe(true) + }) + + it('resets on a new day', async () => { + const db = fakeD1() + await bumpCounter(db, 'ip:abc', 1, DAY) + expect(await bumpCounter(db, 'ip:abc', 1, DAY)).toBe(false) + expect(await bumpCounter(db, 'ip:abc', 1, '2026-08-15')).toBe(true) + }) +}) + +describe('underLimit', () => { + it('is true when nothing has been counted yet', async () => { + const db = fakeD1() + expect(await underLimit(db, 'ip:abc', 3, DAY)).toBe(true) + }) + + it('does not consume anything', async () => { + const db = fakeD1() + await underLimit(db, 'ip:abc', 1, DAY) + await underLimit(db, 'ip:abc', 1, DAY) + // Still gets its full allowance. + expect(await bumpCounter(db, 'ip:abc', 1, DAY)).toBe(true) + }) + + it('goes false at the cap, matching bumpCounter', async () => { + const db = fakeD1() + await bumpCounter(db, 'ip:abc', 2, DAY) + expect(await underLimit(db, 'ip:abc', 2, DAY)).toBe(true) + await bumpCounter(db, 'ip:abc', 2, DAY) + expect(await underLimit(db, 'ip:abc', 2, DAY)).toBe(false) + }) +}) diff --git a/lib/tools/counters.ts b/lib/tools/counters.ts new file mode 100644 index 0000000..3188b05 --- /dev/null +++ b/lib/tools/counters.ts @@ -0,0 +1,105 @@ +/** + * Daily caps for the free tools. + * + * Every cap is one D1 statement. Read-then-increment races across concurrent + * Workers; a single conditional statement does not, because D1 serialises + * writes. This is why there is no Durable Object here — see + * docs/ai-chatbot-architecture.md §3. + * + * The source spec used `UPDATE ... WHERE n < ?`, which affects zero rows when + * today's row does not exist yet. Zero rows means "capped", so every counter + * reported exhausted on the first request after midnight, every day. The + * INSERT ... ON CONFLICT form below is correct on a cold day and keeps the + * single-statement atomicity. + */ + +/** UTC day key, `YYYY-MM-DD`. UTC so the reset time never moves with DST. */ +export function utcDay(now: Date = new Date()): string { + return now.toISOString().slice(0, 10) +} + +/** + * Consume one unit of `key`'s daily allowance. + * + * Returns `true` if it was consumed, `false` if the cap is already reached. + * Allows exactly `limit` calls per UTC day. + */ +export async function bumpCounter( + db: D1Database, + key: string, + limit: number, + day: string = utcDay(), +): Promise { + const res = await db + .prepare( + `INSERT INTO tool_counters (day, key, n) VALUES (?, ?, 1) + ON CONFLICT(day, key) DO UPDATE SET n = n + 1 WHERE n < ?`, + ) + .bind(day, key, limit) + .run() + + return (res.meta.changes ?? 0) > 0 +} + +/** + * Is `key` still under its cap, without consuming anything? + * + * For the crawl route, which checks before doing 20 seconds of work and only + * consumes once a session actually exists — so a typo'd URL does not burn one + * of the visitor's three daily sessions. Check-then-consume is not atomic; the + * failure mode is one extra crawl under a race, which is the right way to be + * wrong here. + */ +export async function underLimit( + db: D1Database, + key: string, + limit: number, + day: string = utcDay(), +): Promise { + const row = await db + .prepare('SELECT n FROM tool_counters WHERE day = ? AND key = ?') + .bind(day, key) + .first<{ n: number }>() + + return (row?.n ?? 0) < limit +} + +/** + * The caps themselves. Phase one runs entirely on free models, so these bound + * abuse and upstream rate limits rather than spend — see docs/ai-chatbot-plan.md. + */ +export const CAPS = { + /** Demo sessions per visitor per day. Key: `ip:`. */ + sessionsPerIp: 3, + /** Messages per day across everyone. Key: `global-messages`. */ + globalMessages: 200, + /** Messages per demo session, then the report. Enforced on the session row. */ + messagesPerSession: 8, +} as const + +/** + * Caps, with an environment override. + * + * Local development shares one counter across every request — there is no + * `CF-Connecting-IP` on localhost, so everything hashes to the same visitor and + * three crawls exhausts the day. Set `TOOLS_SESSIONS_PER_IP` in `.dev.vars` to + * test freely. + * + * Unset in staging and production, so the real limits apply there. A junk value + * falls back to the default rather than disabling the cap — a typo in an env + * var must never quietly turn a limit off. + */ +export type Caps = { -readonly [K in keyof typeof CAPS]: number } + +export function capsFor(env: { TOOLS_SESSIONS_PER_IP?: string }): Caps { + const override = Number(env.TOOLS_SESSIONS_PER_IP) + return { + ...CAPS, + sessionsPerIp: Number.isInteger(override) && override > 0 ? override : CAPS.sessionsPerIp, + } +} + +export const counterKeys = { + ip: (ipHash: string) => `ip:${ipHash}`, + globalMessages: () => 'global-messages', +} as const diff --git a/lib/tools/email.ts b/lib/tools/email.ts new file mode 100644 index 0000000..e2d44f0 --- /dev/null +++ b/lib/tools/email.ts @@ -0,0 +1,99 @@ +/** + * Sending mail. + * + * No provider is configured yet, so this logs instead of sending. Everything + * around it — code generation, hashing, expiry, attempt limits, abuse caps — + * is real, so the day an account exists this is the only function that + * changes. + * + * When adding one (Resend is the likely pick — a plain fetch, no npm package): + * + * const res = await fetch('https://api.resend.com/emails', { + * method: 'POST', + * headers: { + * authorization: `Bearer ${env.RESEND_API_KEY}`, + * 'content-type': 'application/json', + * }, + * body: JSON.stringify({ from: FROM, to, subject, text }), + * }) + * if (!res.ok) throw new Error(`send failed: ${res.status}`) + * + * That also needs SPF and DKIM records on epyc.in, or the mail is rejected or + * spam-filed. The DNS half is not optional and is the part that gets forgotten. + */ + +export type Email = { + to: string + subject: string + text: string +} + +export type SendResult = { sent: boolean; stubbed: boolean } + +/** + * Deliver an email, or log it while no provider exists. + * + * Never throws for a stubbed send — the caller's flow must work identically + * either way, so that swapping in a provider changes delivery and nothing else. + */ +export async function sendEmail( + env: { RESEND_API_KEY?: string }, + email: Email, +): Promise { + if (!env.RESEND_API_KEY) { + // Deliberately readable in `pnpm dev` output: this is how anyone tests the + // flow before a provider exists. + console.warn( + [ + '', + '─────────── EMAIL (not sent — no provider configured) ───────────', + `To: ${email.to}`, + `Subject: ${email.subject}`, + '', + email.text, + '─────────────────────────────────────────────────────────────────', + '', + ].join('\n'), + ) + return { sent: false, stubbed: true } + } + + const res = await fetch('https://api.resend.com/emails', { + method: 'POST', + headers: { + authorization: `Bearer ${env.RESEND_API_KEY}`, + 'content-type': 'application/json', + }, + body: JSON.stringify({ + from: 'EPYC ', + to: email.to, + subject: email.subject, + text: email.text, + }), + }) + + if (!res.ok) { + const detail = await res.text().catch(() => '') + throw new Error(`Email send failed: ${res.status} ${detail.slice(0, 200)}`) + } + + return { sent: true, stubbed: false } +} + +/** The verification email. Plain text — it is one number. */ +export function verificationEmail(to: string, code: string, host: string): Email { + return { + to, + subject: `${code} is your EPYC verification code`, + text: [ + `Your verification code is ${code}`, + '', + `Enter it to get the chatbot code for ${host}.`, + 'The code expires in 10 minutes.', + '', + 'If you did not request this, ignore this email.', + '', + '— EPYC', + ].join('\n'), + } +} diff --git a/lib/tools/models.ts b/lib/tools/models.ts new file mode 100644 index 0000000..a2ecfb9 --- /dev/null +++ b/lib/tools/models.ts @@ -0,0 +1,215 @@ +/** + * OpenRouter client for the free tools. + * + * ponytail: a direct fetch rather than the AI SDK. The plan named `ai` + + * `@openrouter/ai-sdk-provider`, and this is a deliberate departure — we use + * one provider, one model family, no tool calls, no attachments, and no + * multi-provider switching, so the provider abstraction has nothing to + * abstract. What it would add is two dependencies and a documented version + * coupling between them. What we need instead is precise control over the + * fallback chain, which is easier to express here than through a wrapper. + * Ceiling: if we ever want tool calls, structured streaming, or a second + * provider, install the SDK and replace this file — the callers only use + * `streamChat` and `completeJson`. + * + * Every tier is free. Pricing and context verified 14 Aug 2026; see + * docs/ai-chatbot-tech.md. + */ + +export type ChatMessage = { role: 'system' | 'user' | 'assistant'; content: string } + +const ENDPOINT = 'https://openrouter.ai/api/v1/chat/completions' + +/** + * Tried in order. A 429 or 503 advances to the next tier. + * + * Lightning first because it is the fastest (~1.2s), which is the metric the + * whole tool lives or dies on. Super second: larger, same 1M window. Ultra + * last: slower (~6s) and a smaller window, but still six times our corpus. + */ +export const FREE_CHAIN = [ + 'nvidia/nemotron-3.5-lightning:free', + 'nvidia/nemotron-3-super-120b-a12b:free', + 'nvidia/nemotron-3-ultra-550b-a55b:free', +] as const + +/** Only used when OPENROUTER_ALLOW_PAID is explicitly 'true'. Off by default. */ +export const PAID_FALLBACK = 'nvidia/nemotron-3.5-lightning' + +/** + * The scoring call is judgement over the whole corpus, once per session rather + * than eight times, so it starts on the larger model. Still free. + */ +export const SCORING_CHAIN = [ + 'nvidia/nemotron-3-super-120b-a12b:free', + 'nvidia/nemotron-3.5-lightning:free', +] as const + +type CallOptions = { + apiKey: string + messages: ChatMessage[] + chain?: readonly string[] + allowPaid?: boolean + signal?: AbortSignal + /** Low or off — the bot answers from supplied text, it does not solve anything. */ + maxTokens?: number + responseFormatJson?: boolean +} + +/** Which tiers to try, in order. */ +function tiers(opts: CallOptions): string[] { + const chain = [...(opts.chain ?? FREE_CHAIN)] + if (opts.allowPaid) chain.push(PAID_FALLBACK) + return chain +} + +/** Rate limited or temporarily unavailable — worth trying the next tier. */ +function shouldFallOver(status: number): boolean { + return status === 429 || status === 502 || status === 503 || status === 504 +} + +async function call(model: string, opts: CallOptions, stream: boolean): Promise { + return fetch(ENDPOINT, { + method: 'POST', + signal: opts.signal, + headers: { + authorization: `Bearer ${opts.apiKey}`, + 'content-type': 'application/json', + // OpenRouter attributes usage to these; they are not secrets. + 'http-referer': 'https://epyc.in', + 'x-title': 'EPYC Website Diagnostic', + }, + body: JSON.stringify({ + model, + messages: opts.messages, + stream, + max_tokens: opts.maxTokens ?? 700, + temperature: 0.2, + // `effort: 'none'` stops reasoning tokens being generated at all. + // Measured against Lightning: with reasoning on, the model streamed its + // entire chain of thought into `content` — "Here's a thinking process: + // 1. Analyze User Input…" — reciting the system prompt back at the + // visitor and then hitting the token ceiling before writing an answer. + // `exclude` is belt-and-braces for any tier where reasoning is mandatory. + // The job here is reading supplied text, not solving anything. + reasoning: { effort: 'none', exclude: true }, + ...(opts.responseFormatJson ? { response_format: { type: 'json_object' } } : {}), + }), + }) +} + +export type StreamResult = { + /** Plain text deltas. */ + stream: ReadableStream + /** Which tier actually served it — logged so we learn how often tier 1 holds. */ + model: string +} + +/** + * Stream a reply, falling down the chain on rate limits. + * + * The fallback happens before any token is emitted: OpenRouter reports a 429 + * on the initial response, so a busy tier never produces a half-written answer + * that we then abandon. + */ +export async function streamChat(opts: CallOptions): Promise { + let lastStatus = 0 + + for (const model of tiers(opts)) { + const res = await call(model, opts, true) + + if (res.ok && res.body) { + return { stream: toTextStream(res.body), model } + } + + lastStatus = res.status + // Read and discard the error body so the connection is released. + await res.text().catch(() => {}) + if (!shouldFallOver(res.status)) break + } + + throw new Error(`No model available (last status ${lastStatus})`) +} + +/** One-shot JSON response, for the report's scoring call. */ +export async function completeJson(opts: CallOptions): Promise { + let lastStatus = 0 + + for (const model of tiers({ ...opts, chain: opts.chain ?? SCORING_CHAIN })) { + // The report's JSON carries ten questions plus five page types — it needs + // more room than a two-sentence chat answer. + const res = await call(model, { ...opts, maxTokens: opts.maxTokens ?? 2000, responseFormatJson: true }, false) + + if (res.ok) { + const body = (await res.json()) as { choices?: { message?: { content?: string } }[] } + const content = body.choices?.[0]?.message?.content + if (!content) throw new Error('Model returned no content') + return JSON.parse(stripFences(content)) as T + } + + lastStatus = res.status + await res.text().catch(() => {}) + if (!shouldFallOver(res.status)) break + } + + throw new Error(`No model available (last status ${lastStatus})`) +} + +/** Models sometimes wrap JSON in a markdown fence despite json_object mode. */ +function stripFences(s: string): string { + const trimmed = s.trim() + if (!trimmed.startsWith('```')) return trimmed + return trimmed.replace(/^```(?:json)?\s*/i, '').replace(/```\s*$/, '') +} + +/** + * OpenRouter streams OpenAI-shaped SSE. Turn it into plain text deltas. + * + * Buffers across chunk boundaries — a single `data:` line is not guaranteed to + * arrive whole, and splitting naively drops tokens under load. + */ +function toTextStream(body: ReadableStream): ReadableStream { + const decoder = new TextDecoder() + let buffer = '' + + return new ReadableStream({ + async start(controller) { + const reader = body.getReader() + try { + for (;;) { + const { done, value } = await reader.read() + if (done) break + buffer += decoder.decode(value, { stream: true }) + + let nl: number + while ((nl = buffer.indexOf('\n')) !== -1) { + const line = buffer.slice(0, nl).trim() + buffer = buffer.slice(nl + 1) + + if (!line.startsWith('data:')) continue + const payload = line.slice(5).trim() + if (payload === '[DONE]') { + controller.close() + return + } + + try { + const parsed = JSON.parse(payload) as { + choices?: { delta?: { content?: string } }[] + } + const delta = parsed.choices?.[0]?.delta?.content + if (delta) controller.enqueue(delta) + } catch { + // A comment or keep-alive line — ignore it rather than fail the stream. + } + } + } + controller.close() + } catch (err) { + controller.error(err) + } finally { + reader.releaseLock() + } + }, + }) +} diff --git a/lib/tools/session.ts b/lib/tools/session.ts new file mode 100644 index 0000000..b524584 --- /dev/null +++ b/lib/tools/session.ts @@ -0,0 +1,245 @@ +/** + * Session records for the free tools: create, store the crawled corpus, and + * reuse a recent crawl of the same host. + * + * Schema: db/migrations/0003_tool_sessions.sql + */ + +import type { CrawledPage } from '@/lib/crawl/fetch-pages' + +/** How long a crawl of a host stays reusable by a later visitor. */ +export const REUSE_WINDOW_MS = 24 * 60 * 60 * 1000 + +export type SessionStatus = 'crawling' | 'ready' | 'empty' | 'failed' + +/** + * HMAC the visitor's IP before it touches storage. + * + * A bare hash is not anonymisation — IPv4 is 2^32 addresses, which is a + * rainbow table someone can build in an afternoon. The salt is a Worker secret + * (`TOOLS_IP_SALT`), so the stored value is only linkable back by us. + */ +export async function hashIp(ip: string, salt: string): Promise { + const enc = new TextEncoder() + const key = await crypto.subtle.importKey( + 'raw', + enc.encode(salt), + { name: 'HMAC', hash: 'SHA-256' }, + false, + ['sign'], + ) + const sig = await crypto.subtle.sign('HMAC', key, enc.encode(ip)) + return [...new Uint8Array(sig)].map((b) => b.toString(16).padStart(2, '0')).join('') +} + +/** + * Which tool a session belongs to. The table is shared across the free tools, + * so this is how rows are told apart — matches the CHECK constraint in + * db/migrations/0003_tool_sessions.sql. + */ +export type ToolName = 'chatbot' | 'grader' | 'llms-txt' + +export async function createSession( + db: D1Database, + input: { + id: string + tool: ToolName + targetUrl: string + host: string + ipHash: string + status: SessionStatus + }, +): Promise { + await db + .prepare( + `INSERT INTO tool_sessions (id, tool, target_url, host, ip_hash, status) + VALUES (?, ?, ?, ?, ?, ?)`, + ) + .bind(input.id, input.tool, input.targetUrl, input.host, input.ipHash, input.status) + .run() +} + +export async function finishSession( + db: D1Database, + id: string, + status: SessionStatus, + pagesCrawled: number, +): Promise { + await db + .prepare('UPDATE tool_sessions SET status = ?, pages_crawled = ? WHERE id = ?') + .bind(status, pagesCrawled, id) + .run() +} + +export async function savePages( + db: D1Database, + sessionId: string, + pages: CrawledPage[], +): Promise { + if (!pages.length) return + + // One batch, one round trip. D1 charges per statement either way, but the + // latency of 20 sequential awaits is what we are avoiding. + await db.batch( + pages.map((p) => + db + .prepare( + `INSERT OR REPLACE INTO tool_pages (session_id, url, title, text, meta_json) + VALUES (?, ?, ?, ?, ?)`, + ) + .bind( + sessionId, + p.url, + p.title, + p.text, + JSON.stringify({ headings: p.headings, wordCount: p.wordCount, isEmpty: p.isEmpty }), + ), + ), + ) +} + +/** + * The most recent usable crawl of this host, if there is one. + * + * Cuts repeat cost, is politer to the prospect's server, and makes a live + * sales demo of the same domain instant. `force` on the route bypasses this — + * the "I fixed something, read it again" button must never get a cached answer. + */ +export async function findRecentCrawl( + db: D1Database, + host: string, + now: number = Date.now(), +): Promise<{ id: string; pagesCrawled: number } | null> { + const cutoff = new Date(now - REUSE_WINDOW_MS).toISOString().replace('T', ' ').slice(0, 19) + + const row = await db + .prepare( + `SELECT id, pages_crawled FROM tool_sessions + WHERE host = ? AND status = 'ready' AND created_at >= ? + ORDER BY created_at DESC LIMIT 1`, + ) + .bind(host, cutoff) + .first<{ id: string; pages_crawled: number }>() + + return row ? { id: row.id, pagesCrawled: row.pages_crawled } : null +} + +/** Copy a previous session's corpus onto a new session. */ +export async function copyPages(db: D1Database, fromId: string, toId: string): Promise { + const res = await db + .prepare( + `INSERT OR REPLACE INTO tool_pages (session_id, url, title, text, meta_json) + SELECT ?, url, title, text, meta_json FROM tool_pages WHERE session_id = ?`, + ) + .bind(toId, fromId) + .run() + + return res.meta.changes ?? 0 +} + +export type SessionRow = { + id: string + host: string + status: SessionStatus + messages_used: number + transcript_json: string | null + diagnosis_json: string | null +} + +export async function getSession(db: D1Database, id: string): Promise { + return db + .prepare( + `SELECT id, host, status, messages_used, transcript_json, diagnosis_json + FROM tool_sessions WHERE id = ?`, + ) + .bind(id) + .first() +} + +export type Turn = { role: 'user' | 'assistant'; content: string } + +export function readTranscript(row: SessionRow): Turn[] { + if (!row.transcript_json) return [] + try { + return JSON.parse(row.transcript_json) as Turn[] + } catch { + return [] + } +} + +/** + * Record one exchange and consume one of the session's messages. + * + * Single statement so the count cannot drift from the transcript: if the write + * fails, neither happened. Storing the transcript tells us which questions + * visitors actually ask, which is the feedback loop for refining the ten. + */ +export async function recordTurn( + db: D1Database, + id: string, + transcript: Turn[], +): Promise { + await db + .prepare( + `UPDATE tool_sessions + SET transcript_json = ?, messages_used = messages_used + 1 + WHERE id = ?`, + ) + .bind(JSON.stringify(transcript), id) + .run() +} + +export type StoredPage = { url: string; title: string; text: string } + +/** The corpus, for the chat prompt and the report. */ +export async function loadPages(db: D1Database, sessionId: string): Promise { + const { results } = await db + .prepare('SELECT url, title, text FROM tool_pages WHERE session_id = ?') + .bind(sessionId) + .all() + + return results ?? [] +} + +export type ScoringPage = StoredPage & { + headings?: { level: number; text: string }[] + wordCount?: number + isEmpty?: boolean +} + +/** + * The corpus plus the structural facts captured during extraction. + * + * Structure and Specificity are scored from these, so they cost nothing beyond + * the read — the crawl already worked them out. + */ +export async function loadPagesForScoring( + db: D1Database, + sessionId: string, +): Promise { + const { results } = await db + .prepare('SELECT url, title, text, meta_json FROM tool_pages WHERE session_id = ?') + .bind(sessionId) + .all() + + return (results ?? []).map((row) => { + let meta: Partial = {} + try { + if (row.meta_json) meta = JSON.parse(row.meta_json) as Partial + } catch { + // Unparseable metadata just means those dimensions score conservatively. + } + return { url: row.url, title: row.title, text: row.text, ...meta } + }) +} + +export async function saveDiagnosis( + db: D1Database, + sessionId: string, + diagnosis: unknown, +): Promise { + await db + .prepare('UPDATE tool_sessions SET diagnosis_json = ? WHERE id = ?') + .bind(JSON.stringify(diagnosis), sessionId) + .run() +} diff --git a/lib/tools/site-checks.test.ts b/lib/tools/site-checks.test.ts new file mode 100644 index 0000000..2e9a3cb --- /dev/null +++ b/lib/tools/site-checks.test.ts @@ -0,0 +1,143 @@ +import { describe, expect, it } from 'vitest' +import { runSiteChecks, type SitePage } from './site-checks' + +/** + * These checks are shared by every tool on the platform, so a change made for + * one can silently break another. The behaviours locked here are the ones that + * were wrong at least once already. + */ + +const page = (over: Partial = {}): SitePage => ({ + url: 'https://example.com/', + title: 'Home', + text: 'We move freight across the UK. '.repeat(20), + headings: [ + { level: 1, text: 'Home' }, + { level: 2, text: 'Services' }, + ], + wordCount: 120, + isEmpty: false, + ...over, +}) + +describe('crawlability', () => { + it('omits sitemap and robots when the crawl was not observed', () => { + // Reporting "robots.txt not present" for a site that has one is inventing + // a finding. When signals are unknown we say nothing about them. + const { crawlability } = runSiteChecks([page()], {}, false) + const labels = crawlability.checks.map((c) => c.label) + + expect(labels).toEqual(['Readable without JavaScript']) + expect(labels).not.toContain('Sitemap') + expect(labels).not.toContain('robots.txt') + }) + + it('reports all three when the crawl was observed', () => { + const { crawlability } = runSiteChecks( + [page()], + { sitemapFound: true, robotsFound: true, robotsBlockedAll: false }, + true, + ) + expect(crawlability.checks.map((c) => c.label)).toEqual([ + 'Sitemap', + 'robots.txt', + 'Readable without JavaScript', + ]) + expect(crawlability.verdict).toBe('pass') + }) + + it('fails when robots blocks everything', () => { + const { crawlability } = runSiteChecks([page()], { robotsBlockedAll: true }, true) + expect(crawlability.checks.find((c) => c.label === 'robots.txt')?.pass).toBe(false) + }) + + it('fails when nothing was readable without JavaScript', () => { + const { crawlability } = runSiteChecks( + [page({ isEmpty: true, wordCount: 3 })], + { sitemapFound: true, robotsFound: true }, + true, + ) + const readable = crawlability.checks.find((c) => c.label.startsWith('Readable')) + expect(readable?.pass).toBe(false) + expect(readable?.detail).toMatch(/JavaScript/) + }) +}) + +describe('structure', () => { + it('passes a site with real nested headings', () => { + expect(runSiteChecks([page(), page()], {}, true).structure.verdict).toBe('pass') + }) + + it('fails div soup', () => { + const soup = [page({ headings: [] }), page({ headings: [] })] + const { structure } = runSiteChecks(soup, {}, true) + expect(structure.verdict).toBe('fail') + expect(structure.evidence[0]).toMatch(/no headings at all/) + }) + + it('names the pages with no headings, so the finding is actionable', () => { + const pages = [page(), page({ url: 'https://example.com/services', headings: [] })] + const { structure } = runSiteChecks(pages, {}, true) + expect(structure.evidence.join(' ')).toContain('/services') + }) +}) + +describe('specificity', () => { + const vague = 'We are a world-class, industry-leading provider of seamless solutions. ' + + it('quotes vague marketing claims from sales pages', () => { + const { specificity } = runSiteChecks( + [page({ text: vague.repeat(3), wordCount: 30 })], + {}, + true, + ) + expect(specificity.examples.length).toBeGreaterThan(0) + expect(specificity.headline).toMatch(/vague claim/) + }) + + it('ignores blog and news pages', () => { + // Regression: scoring epyc.in flagged a blog post that was itself mocking + // empty language, plus article titles from the blog index. A blog post is + // not the company describing itself. + const { specificity } = runSiteChecks( + [ + page({ url: 'https://example.com/blog/why-buzzwords-fail', text: vague.repeat(5) }), + page({ url: 'https://example.com/news/launch', text: vague.repeat(5) }), + page({ text: 'We move 400 tonnes a week from Felixstowe.', wordCount: 40 }), + ], + {}, + true, + ) + expect(specificity.examples).toHaveLength(0) + expect(specificity.verdict).toBe('pass') + }) + + it('falls back to every page when a site is nothing but blog', () => { + // Otherwise a blog-only site scores suspiciously clean. + const { specificity } = runSiteChecks( + [page({ url: 'https://example.com/blog/one', text: vague.repeat(5), wordCount: 50 })], + {}, + true, + ) + expect(specificity.examples.length).toBeGreaterThan(0) + }) + + it('quotes on word boundaries, not mid-word', () => { + // Regression: quotes used to arrive as "…robust solutions, instead of ", + // cut mid-word. These are shown to a site's owner as evidence, and a + // sloppy fragment undercuts the finding it is meant to support. + const text = `${'padding filler '.repeat(20)}${vague}${'trailing filler '.repeat(20)}` + const { specificity } = runSiteChecks([page({ text })], {}, true) + + expect(specificity.examples.length).toBeGreaterThan(0) + + const words = new Set(text.split(/\s+/).filter(Boolean)) + for (const example of specificity.examples) { + const inner = example.quote.replace(/^…/, '').replace(/…$/, '').trim() + const tokens = inner.split(/\s+/) + // Both ends must be whole words that really appear in the source. + expect(words.has(tokens[0])).toBe(true) + expect(words.has(tokens[tokens.length - 1])).toBe(true) + } + }) +}) diff --git a/lib/tools/site-checks.ts b/lib/tools/site-checks.ts new file mode 100644 index 0000000..162f520 --- /dev/null +++ b/lib/tools/site-checks.ts @@ -0,0 +1,260 @@ +/** + * Site quality checks — the parts of a report that are computed directly from + * a crawl, with no model call. + * + * Shared deliberately. The AI chatbot demo uses these as three of its five + * report dimensions; the Website Grader and the llms.txt Generator score the + * same properties. Anything here must stay tool-agnostic: it takes crawled + * pages plus crawl signals and returns findings, and it knows nothing about + * chatbots, buyer questions, or sessions. + * + * Tool-specific scoring composes on top — see lib/tools/chatbot/diagnosis.ts. + */ + +import type { Heading } from '@/lib/crawl/extract' + +export type Verdict = 'pass' | 'weak' | 'fail' + +export type Check = { label: string; pass: boolean; detail: string } + +/** A crawled page, as much of it as scoring needs. */ +export type SitePage = { + url: string + title: string + text: string + headings?: Heading[] + wordCount?: number + isEmpty?: boolean +} + +/** What the crawl observed about the site as a whole. */ +export type CrawlSignals = { + robotsFound?: boolean + robotsBlockedAll?: boolean + sitemapFound?: boolean + unreachable?: boolean + hitDeadline?: boolean +} + +export type SiteChecks = { + structure: { verdict: Verdict; headline: string; evidence: string[] } + crawlability: { verdict: Verdict; headline: string; checks: Check[] } + specificity: { verdict: Verdict; headline: string; examples: { quote: string; url: string }[] } +} + +/* ------------------------------------------------------------- structure */ + +function scoreStructure(pages: SitePage[]): SiteChecks['structure'] { + const total = pages.length + const noHeadings = pages.filter((p) => (p.headings?.length ?? 0) === 0) + const noH1 = pages.filter((p) => !(p.headings ?? []).some((h) => h.level === 1)) + const flat = pages.filter((p) => new Set((p.headings ?? []).map((h) => h.level)).size <= 1) + + const evidence: string[] = [] + if (noHeadings.length) { + evidence.push( + `${noHeadings.length} of ${total} pages have no headings at all — the page is one undifferentiated block`, + ) + for (const p of noHeadings.slice(0, 3)) evidence.push(`No headings: ${pathOf(p.url)}`) + } + if (noH1.length) evidence.push(`${noH1.length} of ${total} pages have no H1`) + if (flat.length && flat.length !== total) { + evidence.push(`${flat.length} of ${total} pages use only one heading level, so nothing is nested`) + } + if (!evidence.length) { + evidence.push(`All ${total} pages use real, nested headings`) + } + + const badRatio = total ? (noHeadings.length + flat.length / 2) / total : 1 + const verdict: Verdict = badRatio > 0.5 ? 'fail' : badRatio > 0.15 ? 'weak' : 'pass' + + return { + verdict, + headline: verdict === 'pass' ? 'Well structured' : verdict === 'weak' ? 'Weak' : 'Mostly unstructured', + evidence, + } +} + +/* ---------------------------------------------------------- crawlability */ + +function scoreCrawlability( + pages: SitePage[], + signals: CrawlSignals, + signalsKnown: boolean, +): SiteChecks['crawlability'] { + const readable = pages.filter((p) => !p.isEmpty) + + // Readability is derived from the stored pages, so it is always knowable. + const checks: Check[] = [ + { + label: 'Readable without JavaScript', + pass: readable.length > 0 && readable.length >= pages.length / 2, + detail: + readable.length === 0 + ? 'No readable text at all — the pages are built entirely by JavaScript' + : `${readable.length} of ${pages.length} pages returned real text`, + }, + ] + + // Sitemap and robots are only knowable from the crawl itself. Reporting + // "not present" when we simply did not record it would be inventing a + // finding, which is the one thing this report must never do. + if (signalsKnown) { + checks.unshift( + { + label: 'Sitemap', + pass: Boolean(signals.sitemapFound), + detail: signals.sitemapFound + ? 'Found — we used it to pick which pages to read' + : 'Not found, so we had to guess which pages matter by following links', + }, + { + label: 'robots.txt', + pass: !signals.robotsBlockedAll, + detail: signals.robotsBlockedAll + ? 'Blocks automated readers from every page' + : signals.robotsFound + ? 'Present, and allows crawling' + : 'Not present — nothing is blocked, but nothing is directed either', + }, + ) + } + + const failed = checks.filter((c) => !c.pass).length + const verdict: Verdict = failed === 0 ? 'pass' : failed === 1 ? 'weak' : 'fail' + + return { + verdict, + headline: failed === 0 ? 'Passes' : `${failed} of ${checks.length} checks failed`, + checks, + } +} + +/* ----------------------------------------------------------- specificity */ + +/** + * Marketing language that asserts quality without evidence. A claim only + * counts against a site when nothing nearby substantiates it, so each hit is + * quoted in context and the reader can judge. + * + * ponytail: a phrase list, not a model call. Ceiling: it finds stock phrases, + * not every vague sentence, and it cannot tell a substantiated "award-winning" + * from an empty one. If the quoted examples read weak against real sites, this + * moves into the scoring model call as one extra field — same call, no extra + * cost. Decide on real output, per docs/ai-chatbot-plan.md. + */ +const VAGUE = [ + 'world-class', 'world class', 'industry-leading', 'industry leading', 'best-in-class', + 'best in class', 'cutting-edge', 'cutting edge', 'state-of-the-art', 'state of the art', + 'seamless', 'robust', 'innovative', 'passionate', 'one-stop', 'trusted partner', + 'end-to-end', 'tailored solutions', 'bespoke solutions', 'unparalleled', 'holistic', + 'game-changing', 'next-level', 'revolutionary', 'leading provider', 'award-winning', + 'unrivalled', 'unrivaled', 'second to none', 'market-leading', 'best possible', +] + +/** + * Editorial URLs. Excluded from Specificity because the question this + * dimension asks is "does this company describe itself concretely" — and a + * blog post is not the company describing itself. + * + * Measured against real output: scoring epyc.in flagged a blog post that was + * itself *mocking* empty language ("ends with a slide that says 'now go be + * innovative'") as a vague claim, alongside article titles. Every quote we + * show is meant to be evidence the owner cannot argue with; a false positive + * from a blog post hands them the argument. + */ +const EDITORIAL = /\/(blog|news|articles?|insights?|resources?|guides?|press|case-stud)/i + +function scoreSpecificity(pages: SitePage[]): SiteChecks['specificity'] { + const examples: { quote: string; url: string }[] = [] + let hits = 0 + + const salesPages = pages.filter((p) => !EDITORIAL.test(p.url)) + // A site that is nothing but blog has no sales copy to judge; fall back to + // everything rather than reporting a suspiciously clean score. + const judged = salesPages.length > 0 ? salesPages : pages + + for (const page of judged) { + const text = page.text ?? '' + const lower = text.toLowerCase() + + for (const phrase of VAGUE) { + let from = 0 + for (;;) { + const at = lower.indexOf(phrase, from) + if (at === -1) break + hits++ + from = at + phrase.length + + if (examples.length < 6) { + examples.push({ quote: quoteAround(text, at, phrase.length), url: page.url }) + } + } + } + } + + // Density matters more than raw count — a 20-page site will say more of + // everything. Measured over the pages we actually judged, not all of them. + const words = judged.reduce((n, p) => n + (p.wordCount ?? 0), 0) || 1 + const per1k = (hits / words) * 1000 + const verdict: Verdict = per1k > 1.2 ? 'fail' : per1k > 0.4 ? 'weak' : 'pass' + + return { + verdict, + headline: hits === 0 ? 'Concrete throughout' : `${hits} vague claim${hits === 1 ? '' : 's'}`, + examples, + } +} + +/** + * A readable fragment around a match, so the reader sees it in context. + * + * Snaps to word boundaries. These quotes are shown to the site's owner as + * evidence, and a fragment ending "…robust solutions, instead of " reads as + * sloppy rather than damning — which undercuts the finding it is meant to + * support. + */ +function quoteAround(text: string, at: number, length: number): string { + let start = Math.max(0, at - 60) + let end = Math.min(text.length, at + length + 60) + + if (start > 0) { + const space = text.indexOf(' ', start) + if (space !== -1 && space < at) start = space + 1 + } + if (end < text.length) { + const space = text.lastIndexOf(' ', end) + if (space > at + length) end = space + } + + const prefix = start > 0 ? '…' : '' + const suffix = end < text.length ? '…' : '' + return `${prefix}${text.slice(start, end).trim()}${suffix}` +} + +/** + * Run every check that needs no model call. + * + * `signalsKnown: false` means the caller never saw the crawl — the sitemap and + * robots checks are then omitted rather than guessed, because reporting an + * absence we cannot vouch for is the one thing these reports must never do. + */ +export function runSiteChecks( + pages: SitePage[], + signals: CrawlSignals, + signalsKnown = true, +): SiteChecks { + return { + structure: scoreStructure(pages), + crawlability: scoreCrawlability(pages, signals, signalsKnown), + specificity: scoreSpecificity(pages), + } +} + +function pathOf(url: string): string { + try { + return new URL(url).pathname + } catch { + return url + } +} diff --git a/lib/tools/sse-client.ts b/lib/tools/sse-client.ts new file mode 100644 index 0000000..2225f41 --- /dev/null +++ b/lib/tools/sse-client.ts @@ -0,0 +1,45 @@ +/** + * Minimal client-side reader for our SSE routes. + * + * `EventSource` cannot POST, and both tool routes need a request body, so this + * reads the response stream directly. Buffers across chunk boundaries — a + * `data:` line is not guaranteed to arrive whole, and splitting naively drops + * events under load. + */ +export async function readSSE( + res: Response, + onEvent: (event: string, data: Record) => void, +): Promise { + const reader = res.body?.getReader() + if (!reader) return + + const decoder = new TextDecoder() + let buffer = '' + + for (;;) { + const { done, value } = await reader.read() + if (done) break + buffer += decoder.decode(value, { stream: true }) + + // Events are separated by a blank line. + let split: number + while ((split = buffer.indexOf('\n\n')) !== -1) { + const raw = buffer.slice(0, split) + buffer = buffer.slice(split + 2) + + let event = 'message' + let data = '' + for (const line of raw.split('\n')) { + if (line.startsWith('event:')) event = line.slice(6).trim() + else if (line.startsWith('data:')) data += line.slice(5).trim() + } + + if (!data) continue + try { + onEvent(event, JSON.parse(data) as Record) + } catch { + // Malformed frame — skip it rather than kill the stream. + } + } + } +} diff --git a/package.json b/package.json index 3730809..48e9043 100644 --- a/package.json +++ b/package.json @@ -7,6 +7,7 @@ "build": "next build", "start": "next start", "lint": "eslint", + "test": "vitest run", "preview": "opennextjs-cloudflare build && opennextjs-cloudflare preview", "deploy:staging": "NEXT_PUBLIC_DEPLOY_ENV=staging opennextjs-cloudflare build && wrangler deploy --env staging && wrangler deploy --env staging --config workers/contact-webhook/wrangler.jsonc", "deploy:production": "NEXT_PUBLIC_DEPLOY_ENV=production opennextjs-cloudflare build && wrangler deploy --env production && wrangler deploy --env production --config workers/contact-webhook/wrangler.jsonc", @@ -28,7 +29,7 @@ "@cloudflare/workers-types": "^4.20260515.1", "@opennextjs/cloudflare": "^1.19.10", "@tailwindcss/postcss": "^4", - "@types/node": "^20", + "@types/node": "^22.20.1", "@types/react": "^19", "@types/react-dom": "^19", "eslint": "^9", @@ -36,6 +37,7 @@ "tailwindcss": "^4", "tsx": "^4.22.0", "typescript": "^5", + "vitest": "^4.1.10", "wrangler": "^4.91.0" } } diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index a8dc1b3..f0dd145 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -52,8 +52,8 @@ importers: specifier: ^4 version: 4.3.0 '@types/node': - specifier: ^20 - version: 20.19.41 + specifier: ^22.20.1 + version: 22.20.1 '@types/react': specifier: ^19 version: 19.2.14 @@ -75,6 +75,9 @@ importers: typescript: specifier: ^5 version: 5.9.3 + vitest: + specifier: ^4.1.10 + version: 4.1.10(@types/node@22.20.1)(vite@8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0)) wrangler: specifier: ^4.91.0 version: 4.91.0(@cloudflare/workers-types@4.20260515.1) @@ -102,28 +105,24 @@ packages: engines: {node: '>= 10'} cpu: [arm64] os: [linux] - libc: [glibc] '@ast-grep/napi-linux-arm64-musl@0.40.5': resolution: {integrity: sha512-/qKsmds5FMoaEj6FdNzepbmLMtlFuBLdrAn9GIWCqOIcVcYvM1Nka8+mncfeXB/MFZKOrzQsQdPTWqrrQzXLrA==} engines: {node: '>= 10'} cpu: [arm64] os: [linux] - libc: [musl] '@ast-grep/napi-linux-x64-gnu@0.40.5': resolution: {integrity: sha512-DP4oDbq7f/1A2hRTFLhJfDFR6aI5mRWdEfKfHzRItmlKsR9WlcEl1qDJs/zX9R2EEtIDsSKRzuJNfJllY3/W8Q==} engines: {node: '>= 10'} cpu: [x64] os: [linux] - libc: [glibc] '@ast-grep/napi-linux-x64-musl@0.40.5': resolution: {integrity: sha512-BRZUvVBPUNpWPo6Ns8chXVzxHPY+k9gpsubGTHy92Q26ecZULd/dTkWWdnvfhRqttsSQ9Pe/XQdi5+hDQ6RYcg==} engines: {node: '>= 10'} cpu: [x64] os: [linux] - libc: [musl] '@ast-grep/napi-win32-arm64-msvc@0.40.5': resolution: {integrity: sha512-y95zSEwc7vhxmcrcH0GnK4ZHEBQrmrszRBNQovzaciF9GUqEcCACNLoBesn4V47IaOp4fYgD2/EhGRTIBFb2Ug==} @@ -728,105 +727,89 @@ packages: resolution: {integrity: sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw==} cpu: [arm64] os: [linux] - libc: [glibc] '@img/sharp-libvips-linux-arm@1.2.4': resolution: {integrity: sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A==} cpu: [arm] os: [linux] - libc: [glibc] '@img/sharp-libvips-linux-ppc64@1.2.4': resolution: {integrity: sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA==} cpu: [ppc64] os: [linux] - libc: [glibc] '@img/sharp-libvips-linux-riscv64@1.2.4': resolution: {integrity: sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA==} cpu: [riscv64] os: [linux] - libc: [glibc] '@img/sharp-libvips-linux-s390x@1.2.4': resolution: {integrity: sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ==} cpu: [s390x] os: [linux] - libc: [glibc] '@img/sharp-libvips-linux-x64@1.2.4': resolution: {integrity: sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw==} cpu: [x64] os: [linux] - libc: [glibc] '@img/sharp-libvips-linuxmusl-arm64@1.2.4': resolution: {integrity: sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw==} cpu: [arm64] os: [linux] - libc: [musl] '@img/sharp-libvips-linuxmusl-x64@1.2.4': resolution: {integrity: sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg==} cpu: [x64] os: [linux] - libc: [musl] '@img/sharp-linux-arm64@0.34.5': resolution: {integrity: sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [arm64] os: [linux] - libc: [glibc] '@img/sharp-linux-arm@0.34.5': resolution: {integrity: sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [arm] os: [linux] - libc: [glibc] '@img/sharp-linux-ppc64@0.34.5': resolution: {integrity: sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [ppc64] os: [linux] - libc: [glibc] '@img/sharp-linux-riscv64@0.34.5': resolution: {integrity: sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [riscv64] os: [linux] - libc: [glibc] '@img/sharp-linux-s390x@0.34.5': resolution: {integrity: sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [s390x] os: [linux] - libc: [glibc] '@img/sharp-linux-x64@0.34.5': resolution: {integrity: sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [x64] os: [linux] - libc: [glibc] '@img/sharp-linuxmusl-arm64@0.34.5': resolution: {integrity: sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [arm64] os: [linux] - libc: [musl] '@img/sharp-linuxmusl-x64@0.34.5': resolution: {integrity: sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [x64] os: [linux] - libc: [musl] '@img/sharp-wasm32@0.34.5': resolution: {integrity: sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw==} @@ -903,28 +886,24 @@ packages: engines: {node: '>= 10'} cpu: [arm64] os: [linux] - libc: [glibc] '@next/swc-linux-arm64-musl@16.2.6': resolution: {integrity: sha512-URUTu1+dMkxJsPFgm+OeEvq9wf5sujw0EvgYy80TDGHTSLTnIHeqb0Eu8A3sC95IRgjejQL+kC4mw+4yPxiAXA==} engines: {node: '>= 10'} cpu: [arm64] os: [linux] - libc: [musl] '@next/swc-linux-x64-gnu@16.2.6': resolution: {integrity: sha512-DOj182mPV8G3UkrayLoREM5YEYI+Dk5wv7Ox9xl1fFibAELEsFD0lDPfHIeILlutMMfdyhlzYPELG3peuKaurw==} engines: {node: '>= 10'} cpu: [x64] os: [linux] - libc: [glibc] '@next/swc-linux-x64-musl@16.2.6': resolution: {integrity: sha512-HKQ5SP/V/ub73UvF7n/zeJlxk2kLmtL7Wzrg4WfmkjmNos5onJ2tKu7yZOPdL18A6Svfn3max29ym+ry7NkK4g==} engines: {node: '>= 10'} cpu: [x64] os: [linux] - libc: [musl] '@next/swc-win32-arm64-msvc@16.2.6': resolution: {integrity: sha512-LZXpTlPyS5v7HhSmnvsLGP3iIYgYOBnc8r8ArlT55sGHV89bR2HlDdBjWQ+PY6SJMmk8TuVGFuxalnP3k/0Dwg==} @@ -994,6 +973,9 @@ packages: next: '>=15.5.18 <16 || >=16.2.6' wrangler: ^4.86.0 + '@oxc-project/types@0.144.0': + resolution: {integrity: sha512-nuhZIOLuI6TFQ32I/WnUx+SCPY7SdSKwgnFHydAuoS1+Z4BRcaP+RRJmGzl9lw+0OFF7UmaESf7KQRXaNLHypg==} + '@poppinss/colors@4.1.6': resolution: {integrity: sha512-H9xkIdFswbS8n1d6vmRd8+c10t2Qe+rZITbbDHHkQixH5+2x1FDGmi/0K+WgWiqQFKPSlIYB7jlH6Kpfn6Fleg==} @@ -1003,6 +985,93 @@ packages: '@poppinss/exception@1.2.3': resolution: {integrity: sha512-dCED+QRChTVatE9ibtoaxc+WkdzOSjYTKi/+uacHWIsfodVfpsueo3+DKpgU5Px8qXjgmXkSvhXvSCz3fnP9lw==} + '@rolldown/binding-android-arm64@1.2.4': + resolution: {integrity: sha512-jHC2cnyKz5xU2fhECtFl8OZ83cYNt13GZQD+0uMJ/X3o+ijmd56okHhTUwxVSHPx1IRVIJEZ1/1pPzeLCU6XKA==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [android] + + '@rolldown/binding-darwin-arm64@1.2.4': + resolution: {integrity: sha512-Dc5mPD8F5F/FS8i01syd7FTF6yB2fVthH/TRkjwJkzUK6EpoxHtqvZQP5Zwq80/5z19TWYHIg1KOHboCgVx/aQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [darwin] + + '@rolldown/binding-darwin-x64@1.2.4': + resolution: {integrity: sha512-fpDm4oBo6SqLvWUYCmFhdde3U9KH2fRNNMeAnAPAIwxRL345xutL0EtEUcuoxsoazdJGv/MuDBQHlCDrtbvqOg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [darwin] + + '@rolldown/binding-freebsd-x64@1.2.4': + resolution: {integrity: sha512-rSJoreDE/HoIzoaib6MTp5jQtCTdMHKIvItAKT/ImS6Y6Ww76oUaeMyp4Vc/fAgd/ehji068IxetHXAnqUwN9A==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [freebsd] + + '@rolldown/binding-linux-arm-gnueabihf@1.2.4': + resolution: {integrity: sha512-/jm8OGHgn7oGaJu3i/qZI9spUGcJ+y/lk43ttQ/iO1tOd9NissG6o97bighBCiL+BKRngmcDuR6ikfwYdJmVuQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm] + os: [linux] + + '@rolldown/binding-linux-arm64-gnu@1.2.4': + resolution: {integrity: sha512-tIP06BeD9EqvECBrPZ+sqdPlYrT+aYaAiu1wYziVx5elRK/ftm33JxVDy2bXGbr6J0CrtirCkR87/X5a2euEng==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + + '@rolldown/binding-linux-arm64-musl@1.2.4': + resolution: {integrity: sha512-Ql1Q0EQqVThvn9VAVlwNzsUvbSFtCMGjLpRRi4pk5i7NZZ4n5ISiLMjHYtus4VQ2PvkSw24zyaCVsiS+sXPj1w==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + + '@rolldown/binding-linux-ppc64-gnu@1.2.4': + resolution: {integrity: sha512-GjbjXD4XXfN19D0LZNbmiCBUoDiRACsYHr0yaIbbn8aFsXjHZifcYqu/W5Er5X2X990WjHXFrxarn5chzItorQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [ppc64] + os: [linux] + + '@rolldown/binding-linux-s390x-gnu@1.2.4': + resolution: {integrity: sha512-p5WR0NOwaRmJ/B1b6IjEFLLivwEsf3PrdBIhRbhTCQisbo2SvHHpG4ELB/+FgQNnB88LTOF86upmJmbvZdQ2lw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [s390x] + os: [linux] + + '@rolldown/binding-linux-x64-gnu@1.2.4': + resolution: {integrity: sha512-4/GyVjmhR+Tc6HLJvwc1sOhPqAZtySiSMesOZyX6JQ5XBxoTDEMKQzvo07NIK6nTon/SivlZqvhzvuVBNQhObQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + + '@rolldown/binding-linux-x64-musl@1.2.4': + resolution: {integrity: sha512-l9eeLsCNvPpmSXUej0etw/J1eqV0Jj1D5G/xG6YTijmE6dkv6E2QezgWbTfQk63v952DPqrjOCoiqxq7Bw0YUQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + + '@rolldown/binding-openharmony-arm64@1.2.4': + resolution: {integrity: sha512-e0F355MSTMm3+UOqtV3L24gFUp2N5m1f8L/7d56deik6va+AXdrt9F8LbzGpeWGWRbZEDq4m8NVnJDeBtf9DZg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [openharmony] + + '@rolldown/binding-win32-arm64-msvc@1.2.4': + resolution: {integrity: sha512-AWLi0uBRYh6QlE7OKhiz+phZC0qwtij2QZmhmOdsLdFn64m7oMpooE9ICE3lhm9xMb4SpDo2WbHcxX1iFLFtqw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [win32] + + '@rolldown/binding-win32-x64-msvc@1.2.4': + resolution: {integrity: sha512-UwSDJOg3dqCAejWdxclJjCsh3Qq4vLYMDxmyHqo1btz3stK2VqgwNd3mm5tuIwzSlGIQ/1H9Hr+Zn09mrezNqQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [win32] + + '@rolldown/pluginutils@1.0.1': + resolution: {integrity: sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw==} + '@rtsao/scc@1.1.0': resolution: {integrity: sha512-zt6OdqaDoOnJ1ZYsCYGt9YmWzDXl4vQdKTyJev62gFhRGKdx7mcT54V9KIjg+d2wi9EXsPvAPKe7i7WjfVWB8g==} @@ -1165,6 +1234,9 @@ packages: '@speed-highlight/core@1.2.15': resolution: {integrity: sha512-BMq1K3DsElxDWawkX6eLg9+CKJrTVGCBAWVuHXVUV2u0s2711qiChLSId6ikYPfxhdYocLNt3wWwSvDiTvFabw==} + '@standard-schema/spec@1.1.0': + resolution: {integrity: sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==} + '@standard-schema/utils@0.3.0': resolution: {integrity: sha512-e7Mew686owMaPJVNNLs55PUvgz371nKgwsc4vxE49zsODpJEnxgxRo2y/OKrqueavXgZNMDVj3DdHFlaSAeU8g==} @@ -1209,28 +1281,24 @@ packages: engines: {node: '>= 20'} cpu: [arm64] os: [linux] - libc: [glibc] '@tailwindcss/oxide-linux-arm64-musl@4.3.0': resolution: {integrity: sha512-Z6sukiQsngnWO+l39X4pPbiWT81IC+PLKF+PHxIlyZbGNb9MODfYlXEVlFvej5BOZInWX01kVyzeLvHsXhfczQ==} engines: {node: '>= 20'} cpu: [arm64] os: [linux] - libc: [musl] '@tailwindcss/oxide-linux-x64-gnu@4.3.0': resolution: {integrity: sha512-DRNdQRpSGzRGfARVuVkxvM8Q12nh19l4BF/G7zGA1oe+9wcC6saFBHTISrpIcKzhiXtSrlSrluCfvMuledoCTQ==} engines: {node: '>= 20'} cpu: [x64] os: [linux] - libc: [glibc] '@tailwindcss/oxide-linux-x64-musl@4.3.0': resolution: {integrity: sha512-Z0IADbDo8bh6I7h2IQMx601AdXBLfFpEdUotft86evd/8ZPflZe9COPO8Q1vw+pfLWIUo9zN/JGZvwuAJqduqg==} engines: {node: '>= 20'} cpu: [x64] os: [linux] - libc: [musl] '@tailwindcss/oxide-wasm32-wasi@4.3.0': resolution: {integrity: sha512-HNZGOUxEmElksYR7S6sC5jTeNGpobAsy9u7Gu0AskJ8/20FR9GqebUyB+HBcU/ax6BHuiuJi+Oda4B+YX6H1yA==} @@ -1269,6 +1337,12 @@ packages: '@tybys/wasm-util@0.10.2': resolution: {integrity: sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg==} + '@types/chai@5.2.3': + resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} + + '@types/deep-eql@4.0.2': + resolution: {integrity: sha512-c9h9dVVMigMPc4bwTvC5dxqtqJZwQPePsWjPlpSOnojbor6pGqdk541lfA7AqFQr5pB1BRdq0juY9db81BwyFw==} + '@types/estree@1.0.9': resolution: {integrity: sha512-GhdPgy1el4/ImP05X05Uw4cw2/M93BCUmnEvWZNStlCzEKME4Fkk+YpoA5OiHNQmoS7Cafb8Xa3Pya8m1Qrzeg==} @@ -1287,6 +1361,9 @@ packages: '@types/node@20.19.41': resolution: {integrity: sha512-ECymXOukMnOoVkC2bb1Vc/w/836DXncOg5m8Xj1RH7xSHZJWNYY6Zh7EH477vcnD5egKNNfy2RpNOmuChhFPgQ==} + '@types/node@22.20.1': + resolution: {integrity: sha512-EANqOCF9QFyra+4pfxUcX9STKJpCLjMbObVzljIJomAWSnuSIEAvyzEU53GaajbXJEgdh0iEcPL+DGvpUd4k1Q==} + '@types/react-dom@19.2.3': resolution: {integrity: sha512-jp2L/eY6fn+KgVVQAOqYItbF0VY/YApe5Mz2F0aykSO8gx31bYCZyvSeYxCHKvzHG5eZjc+zyaS5BrBWya2+kQ==} peerDependencies: @@ -1393,49 +1470,41 @@ packages: resolution: {integrity: sha512-34gw7PjDGB9JgePJEmhEqBhWvCiiWCuXsL9hYphDF7crW7UgI05gyBAi6MF58uGcMOiOqSJ2ybEeCvHcq0BCmQ==} cpu: [arm64] os: [linux] - libc: [glibc] '@unrs/resolver-binding-linux-arm64-musl@1.11.1': resolution: {integrity: sha512-RyMIx6Uf53hhOtJDIamSbTskA99sPHS96wxVE/bJtePJJtpdKGXO1wY90oRdXuYOGOTuqjT8ACccMc4K6QmT3w==} cpu: [arm64] os: [linux] - libc: [musl] '@unrs/resolver-binding-linux-ppc64-gnu@1.11.1': resolution: {integrity: sha512-D8Vae74A4/a+mZH0FbOkFJL9DSK2R6TFPC9M+jCWYia/q2einCubX10pecpDiTmkJVUH+y8K3BZClycD8nCShA==} cpu: [ppc64] os: [linux] - libc: [glibc] '@unrs/resolver-binding-linux-riscv64-gnu@1.11.1': resolution: {integrity: sha512-frxL4OrzOWVVsOc96+V3aqTIQl1O2TjgExV4EKgRY09AJ9leZpEg8Ak9phadbuX0BA4k8U5qtvMSQQGGmaJqcQ==} cpu: [riscv64] os: [linux] - libc: [glibc] '@unrs/resolver-binding-linux-riscv64-musl@1.11.1': resolution: {integrity: sha512-mJ5vuDaIZ+l/acv01sHoXfpnyrNKOk/3aDoEdLO/Xtn9HuZlDD6jKxHlkN8ZhWyLJsRBxfv9GYM2utQ1SChKew==} cpu: [riscv64] os: [linux] - libc: [musl] '@unrs/resolver-binding-linux-s390x-gnu@1.11.1': resolution: {integrity: sha512-kELo8ebBVtb9sA7rMe1Cph4QHreByhaZ2QEADd9NzIQsYNQpt9UkM9iqr2lhGr5afh885d/cB5QeTXSbZHTYPg==} cpu: [s390x] os: [linux] - libc: [glibc] '@unrs/resolver-binding-linux-x64-gnu@1.11.1': resolution: {integrity: sha512-C3ZAHugKgovV5YvAMsxhq0gtXuwESUKc5MhEtjBpLoHPLYM+iuwSj3lflFwK3DPm68660rZ7G8BMcwSro7hD5w==} cpu: [x64] os: [linux] - libc: [glibc] '@unrs/resolver-binding-linux-x64-musl@1.11.1': resolution: {integrity: sha512-rV0YSoyhK2nZ4vEswT/QwqzqQXw5I6CjoaYMOX0TqBlWhojUf8P94mvI7nuJTeaCkkds3QE4+zS8Ko+GdXuZtA==} cpu: [x64] os: [linux] - libc: [musl] '@unrs/resolver-binding-wasm32-wasi@1.11.1': resolution: {integrity: sha512-5u4RkfxJm+Ng7IWgkzi3qrFOvLvQYnPBmjmZQ8+szTK/b31fQCnleNl1GgEt7nIsZRIf5PLhPwT0WM+q45x/UQ==} @@ -1457,6 +1526,35 @@ packages: cpu: [x64] os: [win32] + '@vitest/expect@4.1.10': + resolution: {integrity: sha512-YsCn+qAk1GWjQOWFEsEcL2gNQ0zmVmQu3T03qP6UyjhtmdtwtbuI+DASn/7iQB3HGTXkdBwGddzxPlmiql5vlA==} + + '@vitest/mocker@4.1.10': + resolution: {integrity: sha512-v0xaezt+DKEmKfaxg133ldzADrwLGd7Ze1MfQQTYfvs8OqZIwbxyxaYURivwV7sWy5fqn3rH5uOrSp07bp44Ow==} + peerDependencies: + msw: ^2.4.9 + vite: ^6.0.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + msw: + optional: true + vite: + optional: true + + '@vitest/pretty-format@4.1.10': + resolution: {integrity: sha512-W1HsjSH4MXQ9YfmmhLAoIYf1HRfekQCGngeIgcei6MP5QQGWUe0gkopdZQaVCFO+JDJMrAJGwa5pRpNpvy4P8Q==} + + '@vitest/runner@4.1.10': + resolution: {integrity: sha512-IKI6kpIH+LmpROplyLwBBaCfMgOZOMsygVa6BARD6ahA04VRuJSa6OaVG7kRvSEMD870Vd91rSSw0eegtWyLGg==} + + '@vitest/snapshot@4.1.10': + resolution: {integrity: sha512-xRkfOT1qpTAi/Ti4Y1LtfRc3kEuqxGw59eN2jN9pRWMtS/XDevekhcFSqvQqjUNGksfjMJu3Y+oJ+4Ypn2OaJw==} + + '@vitest/spy@4.1.10': + resolution: {integrity: sha512-PLf/Ugvoq5wO/b4rwYCR1h2PSIdXz7wnkQFMiUpLdtM7l6pqVFcQIBEHyT1+l+cj7mNwAfZHzqXqDyjvOuwbDw==} + + '@vitest/utils@4.1.10': + resolution: {integrity: sha512-fy9am/HWxbaGt/Sawrp90vt6Y6jQwf1RX77cz3uwoJwJVMli/e1IEwRPnMNJ7vKfPTwo0diXifkpPvwH9v7nGA==} + abort-controller@3.0.0: resolution: {integrity: sha512-h8lQ8tacZYnR3vNQTgibj+tODHI5/+l06Au2Pcriv/Gmet0eaj4TwWH41sO9wnHDiQsEj19q0drzdWdeAHtweg==} engines: {node: '>=6.5'} @@ -1544,6 +1642,10 @@ packages: resolution: {integrity: sha512-BNoCY6SXXPQ7gF2opIP4GBE+Xw7U+pHMYKuzjgCN3GwiaIR09UUeKfheyIry77QtrCBlC0KK0q5/TER/tYh3PQ==} engines: {node: '>= 0.4'} + assertion-error@2.0.1: + resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} + engines: {node: '>=12'} + ast-types-flow@0.0.8: resolution: {integrity: sha512-OH/2E5Fg20h2aPrbe+QL8JZQFko0YZaF+j4mnQ7BGhfavO7OpSLa8a0y9sBwomHdSbkhTS8TQNayBfnW5DwbvQ==} @@ -1636,6 +1738,10 @@ packages: caniuse-lite@1.0.30001792: resolution: {integrity: sha512-hVLMUZFgR4JJ6ACt1uEESvQN1/dBVqPAKY0hgrV70eN3391K6juAfTjKZLKvOMsx8PxA7gsY1/tLMMTcfFLLpw==} + chai@6.2.2: + resolution: {integrity: sha512-NUPRluOfOiTKBKvWPtSD4PhFvWCqOi0BGStNWs57X9js7XGTprSmFoz5F0tWhR4WPjNeR9jXqdC7/UpSJTnlRg==} + engines: {node: '>=18'} + chalk@4.1.2: resolution: {integrity: sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==} engines: {node: '>=10'} @@ -1841,6 +1947,9 @@ packages: resolution: {integrity: sha512-HVLACW1TppGYjJ8H6/jqH/pqOtKRw6wMlrB23xfExmFWxFquAIWCmwoLsOyN96K4a5KbmOf5At9ZUO3GZbetAw==} engines: {node: '>= 0.4'} + es-module-lexer@2.3.1: + resolution: {integrity: sha512-shc1dbU90Yl/xq1QrC7QRtfcwURZuVRfPhZbDoldJ1cn1gzDvBaBWlv0eFolj5+0znnPJz5TXLxsN77X/12KTA==} + es-object-atoms@1.1.1: resolution: {integrity: sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA==} engines: {node: '>= 0.4'} @@ -1994,6 +2103,9 @@ packages: resolution: {integrity: sha512-MMdARuVEQziNTeJD8DgMqmhwR11BRQ/cBP+pLtYdSTnf3MIO8fFeiINEbX36ZdNlfU/7A9f3gUw49B3oQsvwBA==} engines: {node: '>=4.0'} + estree-walker@3.0.3: + resolution: {integrity: sha512-7RUKfXgSMMkzt6ZuXmqapOurLGPPfgj6l9uRZ7lRGolvk0y2yocc35LdcxKC5PQZdn2DMqioAQ2NoWcrTKmm6g==} + esutils@2.0.3: resolution: {integrity: sha512-kVscqXk4OCp68SZ0dkgEKVi6/8ij300KBWTJq32P/dYeWTSwK41WyTxalN1eRmA5Z9UU/LX9D7FWSmV9SAYx6g==} engines: {node: '>=0.10.0'} @@ -2010,6 +2122,10 @@ packages: resolution: {integrity: sha512-8uSpZZocAZRBAPIEINJj3Lo9HyGitllczc27Eh5YYojjMFMn8yHMDMaUHE2Jqfq05D/wucwI4JGURyXt1vchyg==} engines: {node: '>=10'} + expect-type@1.4.0: + resolution: {integrity: sha512-KfYbmpRm0VbLjEvVa9yGwCi9GI34xvi7A/HXYWQO65CSD2u3MczUJSuwXKFIxlGsgBQizV9q5J9NHj4VG0n+pA==} + engines: {node: '>=12.0.0'} + express@5.2.1: resolution: {integrity: sha512-hIS4idWWai69NezIdRt2xFVofaF4j+6INOpJlVOLDO8zXGpUVEVzIYk12UUi2JzjEzWL3IOAxcTubgz9Po0yXw==} engines: {node: '>= 18'} @@ -2469,57 +2585,107 @@ packages: cpu: [arm64] os: [android] + lightningcss-android-arm64@1.33.0: + resolution: {integrity: sha512-gEpRTalKdosp4Bb8qWtc2iOgE5SeIHlpS1up9bFq2wAyYhl1UdTObYiHe98zEM9SQvSoqQZ1IQD0JNpg3Ml5pg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [android] + lightningcss-darwin-arm64@1.32.0: resolution: {integrity: sha512-RzeG9Ju5bag2Bv1/lwlVJvBE3q6TtXskdZLLCyfg5pt+HLz9BqlICO7LZM7VHNTTn/5PRhHFBSjk5lc4cmscPQ==} engines: {node: '>= 12.0.0'} cpu: [arm64] os: [darwin] + lightningcss-darwin-arm64@1.33.0: + resolution: {integrity: sha512-Sciaz8eenNTKn9b3t7+xr0ipTp9YxKQY4npwQ3mrRuL0BAVHBLyZxofhaKBAVtzmtRZ/zTyo0/to4B1uWG/Djg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [darwin] + lightningcss-darwin-x64@1.32.0: resolution: {integrity: sha512-U+QsBp2m/s2wqpUYT/6wnlagdZbtZdndSmut/NJqlCcMLTWp5muCrID+K5UJ6jqD2BFshejCYXniPDbNh73V8w==} engines: {node: '>= 12.0.0'} cpu: [x64] os: [darwin] + lightningcss-darwin-x64@1.33.0: + resolution: {integrity: sha512-Z5UPAxzrjlWNNyGy6i65cJzzvgJ5D3T6wMvs+gWpY9d7qRhANrxqAp6LhxIgZhWEw18RfJTGcRxjuLIBr+m8XQ==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [darwin] + lightningcss-freebsd-x64@1.32.0: resolution: {integrity: sha512-JCTigedEksZk3tHTTthnMdVfGf61Fky8Ji2E4YjUTEQX14xiy/lTzXnu1vwiZe3bYe0q+SpsSH/CTeDXK6WHig==} engines: {node: '>= 12.0.0'} cpu: [x64] os: [freebsd] + lightningcss-freebsd-x64@1.33.0: + resolution: {integrity: sha512-QQM/Ti/hQajJwCY+RiWuCZ9sdtI/XQk7nDK5vC8kkdwixezOlDgvDx7+RT+QjK6FcFT4MpsuoBnHIo/O3StRRg==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [freebsd] + lightningcss-linux-arm-gnueabihf@1.32.0: resolution: {integrity: sha512-x6rnnpRa2GL0zQOkt6rts3YDPzduLpWvwAF6EMhXFVZXD4tPrBkEFqzGowzCsIWsPjqSK+tyNEODUBXeeVHSkw==} engines: {node: '>= 12.0.0'} cpu: [arm] os: [linux] + lightningcss-linux-arm-gnueabihf@1.33.0: + resolution: {integrity: sha512-N7FVBe6iS24MlM6R/4RBTxGhQheZGs7tiQ9U32UtF75NzP5Q7xWPRqLBCKxlRQRk3rY1jCIPLzx7WzOhuUIRLQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm] + os: [linux] + lightningcss-linux-arm64-gnu@1.32.0: resolution: {integrity: sha512-0nnMyoyOLRJXfbMOilaSRcLH3Jw5z9HDNGfT/gwCPgaDjnx0i8w7vBzFLFR1f6CMLKF8gVbebmkUN3fa/kQJpQ==} engines: {node: '>= 12.0.0'} cpu: [arm64] os: [linux] - libc: [glibc] + + lightningcss-linux-arm64-gnu@1.33.0: + resolution: {integrity: sha512-j2v/itmy4HlNxlc6voKXYgBqNi0Ng2LShg4z7GufpEgs05P+2suBVyi9I6YHq5uoVFx9ETin3eCEhLVyXGQnKg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] lightningcss-linux-arm64-musl@1.32.0: resolution: {integrity: sha512-UpQkoenr4UJEzgVIYpI80lDFvRmPVg6oqboNHfoH4CQIfNA+HOrZ7Mo7KZP02dC6LjghPQJeBsvXhJod/wnIBg==} engines: {node: '>= 12.0.0'} cpu: [arm64] os: [linux] - libc: [musl] + + lightningcss-linux-arm64-musl@1.33.0: + resolution: {integrity: sha512-yiO5ROMuYQgXbC60yjZU5CYSFZGKXL0HFATXt9mHJn1+zW55oCtMI9NfcVhYLMFDL7gV7oBPon/EmMMGg2OvtQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] lightningcss-linux-x64-gnu@1.32.0: resolution: {integrity: sha512-V7Qr52IhZmdKPVr+Vtw8o+WLsQJYCTd8loIfpDaMRWGUZfBOYEJeyJIkqGIDMZPwPx24pUMfwSxxI8phr/MbOA==} engines: {node: '>= 12.0.0'} cpu: [x64] os: [linux] - libc: [glibc] + + lightningcss-linux-x64-gnu@1.33.0: + resolution: {integrity: sha512-ar+Ju7LmcN0Jo4FpL4hpFybwNG9/3A/Br5KW2n2jyODg3MEZXaDYADdemoNS+BDNfMgKvylJLj4S5tyRActuAg==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] lightningcss-linux-x64-musl@1.32.0: resolution: {integrity: sha512-bYcLp+Vb0awsiXg/80uCRezCYHNg1/l3mt0gzHnWV9XP1W5sKa5/TCdGWaR/zBM2PeF/HbsQv/j2URNOiVuxWg==} engines: {node: '>= 12.0.0'} cpu: [x64] os: [linux] - libc: [musl] + + lightningcss-linux-x64-musl@1.33.0: + resolution: {integrity: sha512-RYiYbkokw0trfKqqzfF55lginwEPrD3OJDfTuJzFs1MK6iFnDenaz1fqLLtX4ITG3OktJQXOeTaw1awrBAlZPw==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] lightningcss-win32-arm64-msvc@1.32.0: resolution: {integrity: sha512-8SbC8BR40pS6baCM8sbtYDSwEVQd4JlFTOlaD3gWGHfThTcABnNDBda6eTZeqbofalIJhFx0qKzgHJmcPTnGdw==} @@ -2527,16 +2693,32 @@ packages: cpu: [arm64] os: [win32] + lightningcss-win32-arm64-msvc@1.33.0: + resolution: {integrity: sha512-1K+MPfLSFVpphzpdbfkhlWk6wBrTObBzS2T6db10PNOZgR9GoVsAWzwNyuhUYYbTp23j+4RrncfujZ4uAzXvwA==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [win32] + lightningcss-win32-x64-msvc@1.32.0: resolution: {integrity: sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q==} engines: {node: '>= 12.0.0'} cpu: [x64] os: [win32] + lightningcss-win32-x64-msvc@1.33.0: + resolution: {integrity: sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [win32] + lightningcss@1.32.0: resolution: {integrity: sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ==} engines: {node: '>= 12.0.0'} + lightningcss@1.33.0: + resolution: {integrity: sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA==} + engines: {node: '>= 12.0.0'} + locate-path@6.0.0: resolution: {integrity: sha512-iPZK6eYjbxRu3uB4/WZ3EsEIMJFMqAoopl3R+zuq0UjcAm/MO6KCweDgPfP3elTztoKP3KtnVHxTn2NHBSDVUw==} engines: {node: '>=10'} @@ -2667,6 +2849,11 @@ packages: engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} hasBin: true + nanoid@3.3.18: + resolution: {integrity: sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w==} + engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} + hasBin: true + napi-postinstall@0.3.4: resolution: {integrity: sha512-PHI5f1O0EP5xJ9gQmFGMS6IZcrVvTjpXjz7Na41gTE7eE2hK11lg04CECCYEEjdc17EV4DO+fkGEtt7TpTaTiQ==} engines: {node: ^12.20.0 || ^14.18.0 || >=16.0.0} @@ -2764,6 +2951,10 @@ packages: obliterator@1.6.1: resolution: {integrity: sha512-9WXswnqINnnhOG/5SLimUlzuU1hFJUc8zkwyD59Sd+dPOMf05PmnYG/d6Q7HZ+KmgkZJa1PxRso6QdM3sTNHig==} + obug@2.1.4: + resolution: {integrity: sha512-4a+OsYv9UktOJKE+l1A4OufDgdRF9PifWj+tJnHURo/P+WOxpG4GzUFL9qCalmWauao6ogiG+QvnCovwPoyAWA==} + engines: {node: '>=12.20.0'} + on-finished@2.4.1: resolution: {integrity: sha512-oVlzkg3ENAhCk2zdv7IJwd/QUD4z2RxRwpkcGY8psCVcCYZNq4wYnVWALHM+brtuJjePWiYF/ClmuDr8Ch5+kg==} engines: {node: '>= 0.8'} @@ -2845,6 +3036,10 @@ packages: resolution: {integrity: sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==} engines: {node: '>=12'} + picomatch@4.0.5: + resolution: {integrity: sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==} + engines: {node: '>=12'} + possible-typed-array-names@1.1.0: resolution: {integrity: sha512-/+5VFTchJDoVj3bhoqi6UeymcD00DAwb1nJwamzPvHEszJ4FpF6SNNbUbOS8yI56qHzdV8eK0qEfOSiodkTdxg==} engines: {node: '>= 0.4'} @@ -2857,6 +3052,10 @@ packages: resolution: {integrity: sha512-SoSL4+OSEtR99LHFZQiJLkT59C5B1amGO1NzTwj7TT1qCUgUO6hxOvzkOYxD+vMrXBM3XJIKzokoERdqQq/Zmg==} engines: {node: ^10 || ^12 || >=14} + postcss@8.5.26: + resolution: {integrity: sha512-u82N74LFzG8ca+dD8puPnplTXoGH4fTPpVGuIbt36G3qvNlkvfD0lEAZSxaly3KX8TS/L1A1gsCEmvKmBcVbkQ==} + engines: {node: ^10 || ^12 || >=14} + prelude-ls@1.2.1: resolution: {integrity: sha512-vkcDPrRZo1QZLbn5RLGPpg/WmIQ65qoWWhcGKf/b5eplkkarX0m9z8ppCat4mlOqUsWpyNuYgO3VRyrYHSzX5g==} engines: {node: '>= 0.8.0'} @@ -2929,6 +3128,11 @@ packages: resolution: {integrity: sha512-g6QUff04oZpHs0eG5p83rFLhHeV00ug/Yf9nZM6fLeUrPguBTkTQOdpAWWspMh55TZfVQDPaN3NQJfbVRAxdIw==} engines: {iojs: '>=1.0.0', node: '>=0.10.0'} + rolldown@1.2.4: + resolution: {integrity: sha512-rSr7irW0K7QRWzjdJXqZowkcRdDtjRduh43rBltnVKd0VFq839l1lJoDvGJb6gl7+4rTTCrPWu+YfujUL8Ug7w==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + router@2.2.0: resolution: {integrity: sha512-nLTrUKm2UyiL7rlhapu/Zl45FwNgkZGaCpZbIHajDYgwlJCOzLSk+cIPAnsEqV955GjILJnKbdQC1nVPz+gAYQ==} engines: {node: '>= 18'} @@ -3014,6 +3218,9 @@ packages: resolution: {integrity: sha512-ZX99e6tRweoUXqR+VBrslhda51Nh5MTQwou5tnUDgbtyM0dBgmhEDtWGP/xbKn6hqfPRHujUNwz5fy/wbbhnpw==} engines: {node: '>= 0.4'} + siginfo@2.0.0: + resolution: {integrity: sha512-ybx0WO1/8bSBLEWXZvEd7gMW3Sn3JFlW3TvX1nREbDLRNQNaeNN8WK0meBwPdAaOI7TtRRRJn/Es1zhrrCHu7g==} + signal-exit@3.0.7: resolution: {integrity: sha512-wnD2ZE+l+SPC/uoS0vXeE9L1+0wuaMqKlfz9AMUo38JsyLSBWSFcHR1Rri62LZc12vLr1gb3jl7iwQhgwpAbGQ==} @@ -3035,10 +3242,16 @@ packages: stable-hash@0.0.5: resolution: {integrity: sha512-+L3ccpzibovGXFK+Ap/f8LOS0ahMrHTf3xu7mMLSpEGU0EO9ucaysSylKo9eRDFNhWve/y275iPmIZ4z39a9iA==} + stackback@0.0.2: + resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + statuses@2.0.2: resolution: {integrity: sha512-DvEy55V3DB7uknRo+4iOGT5fP1slR8wQohVdknigZPMpMstaKJQWhwiYBACJE3Ul2pTnATihhBYnRhZQHGBiRw==} engines: {node: '>= 0.8'} + std-env@4.2.0: + resolution: {integrity: sha512-oCUKSupKTHX53EyjDtuZQ64pjLJ6yYCtpmEw0goYxtjG9KpbRe8KAsl2tBUGU9DyMcJ0RwJ8GqJAFzMXcXW1Rw==} + stop-iteration-iterator@1.1.0: resolution: {integrity: sha512-eLoXW/DHyl62zxY4SCaIgnRhuMr6ri4juEYARS8E6sCEqzKpOiE521Ucofdx+KnDZl5xmvGYaaKCk5FEOxJCoQ==} engines: {node: '>= 0.4'} @@ -3133,10 +3346,25 @@ packages: engines: {node: '>=10'} hasBin: true + tinybench@2.9.0: + resolution: {integrity: sha512-0+DUvqWMValLmha6lr4kD8iAMK1HzV0/aKnCtWb9v9641TnP/MFb7Pc2bxoxQjTXAErryXVgUOfv2YqNllqGeg==} + + tinyexec@1.3.0: + resolution: {integrity: sha512-QKAl9m8gWWGHV8jZcPeym6j+XULi6tOf1mT83WYJ4Lk2ytW/uwAWkrP0uFsdoYMdueVJ0qs26wZ+23xeB4ibNQ==} + engines: {node: '>=18'} + tinyglobby@0.2.16: resolution: {integrity: sha512-pn99VhoACYR8nFHhxqix+uvsbXineAasWm5ojXoN8xEwK5Kd3/TrhNn1wByuD52UxWRLy8pu+kRMniEi6Eq9Zg==} engines: {node: '>=12.0.0'} + tinyglobby@0.2.17: + resolution: {integrity: sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g==} + engines: {node: '>=12.0.0'} + + tinyrainbow@3.1.1: + resolution: {integrity: sha512-yau8yJdTt989Mm0Bd/236QnzEiPf2xLLTqUZRUJOo/3CB078LSwzei343DgtJVmfJKJE3TMINY1u42SQsP6mXw==} + engines: {node: '>=14.0.0'} + to-regex-range@5.0.1: resolution: {integrity: sha512-65P7iz6X5yEr1cwcgvQxbbIw7Uk3gOy5dIdtZ4rDveLqhrdJP+Li/Hx6tyK0NEb+2GCyneCMJiGqrADCSNk8sQ==} engines: {node: '>=8.0'} @@ -3244,6 +3472,90 @@ packages: resolution: {integrity: sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg==} engines: {node: '>= 0.8'} + vite@8.2.1: + resolution: {integrity: sha512-EU/eS7BH3XROHh2YnBefjM6DBKA6ZeMZEYQbj7NLWg5wHYlhB8B/Mayd5XsgWq+NFYccDOTemRpdETWR6Ka/lw==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + peerDependencies: + '@types/node': ^20.19.0 || >=22.12.0 + '@vitejs/devtools': ^0.4.0 + esbuild: ^0.27.0 + jiti: '>=1.21.0' + less: ^4.0.0 + sass: ^1.70.0 + sass-embedded: ^1.70.0 + stylus: '>=0.54.8' + sugarss: ^5.0.0 + terser: ^5.16.0 + tsx: ^4.8.1 + yaml: ^2.4.2 + peerDependenciesMeta: + '@types/node': + optional: true + '@vitejs/devtools': + optional: true + esbuild: + optional: true + jiti: + optional: true + less: + optional: true + sass: + optional: true + sass-embedded: + optional: true + stylus: + optional: true + sugarss: + optional: true + terser: + optional: true + tsx: + optional: true + yaml: + optional: true + + vitest@4.1.10: + resolution: {integrity: sha512-R9jUTe5S4Qb0HCd4TNqpC7oGcrMssMRGXLW80ubjWsW9VH5GF8y1Y0SFLY9AbqSk6nt0PnOx4H4WNJYZ13GUPw==} + engines: {node: ^20.0.0 || ^22.0.0 || >=24.0.0} + hasBin: true + peerDependencies: + '@edge-runtime/vm': '*' + '@opentelemetry/api': ^1.9.0 + '@types/node': ^20.0.0 || ^22.0.0 || >=24.0.0 + '@vitest/browser-playwright': 4.1.10 + '@vitest/browser-preview': 4.1.10 + '@vitest/browser-webdriverio': 4.1.10 + '@vitest/coverage-istanbul': 4.1.10 + '@vitest/coverage-v8': 4.1.10 + '@vitest/ui': 4.1.10 + happy-dom: '*' + jsdom: '*' + vite: ^6.0.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + '@edge-runtime/vm': + optional: true + '@opentelemetry/api': + optional: true + '@types/node': + optional: true + '@vitest/browser-playwright': + optional: true + '@vitest/browser-preview': + optional: true + '@vitest/browser-webdriverio': + optional: true + '@vitest/coverage-istanbul': + optional: true + '@vitest/coverage-v8': + optional: true + '@vitest/ui': + optional: true + happy-dom: + optional: true + jsdom: + optional: true + web-streams-polyfill@4.0.0-beta.3: resolution: {integrity: sha512-QW95TCTaHmsYfHDybGMwO5IJIM93I/6vTRk+daHTWFPhwh+C8Cg7j7XyKrwrj8Ib6vYXe0ocYNrmzY4xAAN6ug==} engines: {node: '>= 14'} @@ -3280,6 +3592,11 @@ packages: engines: {node: ^16.13.0 || >=18.0.0} hasBin: true + why-is-node-running@2.3.0: + resolution: {integrity: sha512-hUrmaWBdVDcxvYqnyh09zunKzROWjbZTiNy8dBEjkS7ehEDQibXJ7XvlmtbwuTclUiIyN+CyXQD4Vmko8fNm8w==} + engines: {node: '>=8'} + hasBin: true + word-wrap@1.2.5: resolution: {integrity: sha512-BN22B5eaMMI9UMtjrGd5g5eCYPpCPDUy0FJXbYsaT5zYxjFOckS53SQDE3pWkVoWpHXVb3BrYcEN4Twa55B5cA==} engines: {node: '>=0.10.0'} @@ -4571,6 +4888,8 @@ snapshots: - encoding - supports-color + '@oxc-project/types@0.144.0': {} + '@poppinss/colors@4.1.6': dependencies: kleur: 4.1.5 @@ -4583,6 +4902,50 @@ snapshots: '@poppinss/exception@1.2.3': {} + '@rolldown/binding-android-arm64@1.2.4': + optional: true + + '@rolldown/binding-darwin-arm64@1.2.4': + optional: true + + '@rolldown/binding-darwin-x64@1.2.4': + optional: true + + '@rolldown/binding-freebsd-x64@1.2.4': + optional: true + + '@rolldown/binding-linux-arm-gnueabihf@1.2.4': + optional: true + + '@rolldown/binding-linux-arm64-gnu@1.2.4': + optional: true + + '@rolldown/binding-linux-arm64-musl@1.2.4': + optional: true + + '@rolldown/binding-linux-ppc64-gnu@1.2.4': + optional: true + + '@rolldown/binding-linux-s390x-gnu@1.2.4': + optional: true + + '@rolldown/binding-linux-x64-gnu@1.2.4': + optional: true + + '@rolldown/binding-linux-x64-musl@1.2.4': + optional: true + + '@rolldown/binding-openharmony-arm64@1.2.4': + optional: true + + '@rolldown/binding-win32-arm64-msvc@1.2.4': + optional: true + + '@rolldown/binding-win32-x64-msvc@1.2.4': + optional: true + + '@rolldown/pluginutils@1.0.1': {} + '@rtsao/scc@1.1.0': {} '@sindresorhus/is@7.2.0': {} @@ -4783,6 +5146,8 @@ snapshots: '@speed-highlight/core@1.2.15': {} + '@standard-schema/spec@1.1.0': {} + '@standard-schema/utils@0.3.0': {} '@swc/helpers@0.5.15': @@ -4865,6 +5230,13 @@ snapshots: tslib: 2.8.1 optional: true + '@types/chai@5.2.3': + dependencies: + '@types/deep-eql': 4.0.2 + assertion-error: 2.0.1 + + '@types/deep-eql@4.0.2': {} + '@types/estree@1.0.9': {} '@types/json-schema@7.0.15': {} @@ -4884,6 +5256,10 @@ snapshots: dependencies: undici-types: 6.21.0 + '@types/node@22.20.1': + dependencies: + undici-types: 6.21.0 + '@types/react-dom@19.2.3(@types/react@19.2.14)': dependencies: '@types/react': 19.2.14 @@ -5042,6 +5418,47 @@ snapshots: '@unrs/resolver-binding-win32-x64-msvc@1.11.1': optional: true + '@vitest/expect@4.1.10': + dependencies: + '@standard-schema/spec': 1.1.0 + '@types/chai': 5.2.3 + '@vitest/spy': 4.1.10 + '@vitest/utils': 4.1.10 + chai: 6.2.2 + tinyrainbow: 3.1.1 + + '@vitest/mocker@4.1.10(vite@8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0))': + dependencies: + '@vitest/spy': 4.1.10 + estree-walker: 3.0.3 + magic-string: 0.30.21 + optionalDependencies: + vite: 8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0) + + '@vitest/pretty-format@4.1.10': + dependencies: + tinyrainbow: 3.1.1 + + '@vitest/runner@4.1.10': + dependencies: + '@vitest/utils': 4.1.10 + pathe: 2.0.3 + + '@vitest/snapshot@4.1.10': + dependencies: + '@vitest/pretty-format': 4.1.10 + '@vitest/utils': 4.1.10 + magic-string: 0.30.21 + pathe: 2.0.3 + + '@vitest/spy@4.1.10': {} + + '@vitest/utils@4.1.10': + dependencies: + '@vitest/pretty-format': 4.1.10 + convert-source-map: 2.0.0 + tinyrainbow: 3.1.1 + abort-controller@3.0.0: dependencies: event-target-shim: 5.0.1 @@ -5153,6 +5570,8 @@ snapshots: get-intrinsic: 1.3.0 is-array-buffer: 3.0.5 + assertion-error@2.0.1: {} + ast-types-flow@0.0.8: {} async-function@1.0.0: {} @@ -5243,6 +5662,8 @@ snapshots: caniuse-lite@1.0.30001792: {} + chai@6.2.2: {} + chalk@4.1.2: dependencies: ansi-styles: 4.3.0 @@ -5492,6 +5913,8 @@ snapshots: iterator.prototype: 1.1.5 math-intrinsics: 1.1.0 + es-module-lexer@2.3.1: {} + es-object-atoms@1.1.1: dependencies: es-errors: 1.3.0 @@ -5753,6 +6176,10 @@ snapshots: estraverse@5.3.0: {} + estree-walker@3.0.3: + dependencies: + '@types/estree': 1.0.9 + esutils@2.0.3: {} etag@1.8.1: {} @@ -5771,6 +6198,8 @@ snapshots: signal-exit: 3.0.7 strip-final-newline: 2.0.0 + expect-type@1.4.0: {} + express@5.2.1: dependencies: accepts: 2.0.0 @@ -5838,6 +6267,10 @@ snapshots: optionalDependencies: picomatch: 4.0.4 + fdir@6.5.0(picomatch@4.0.5): + optionalDependencies: + picomatch: 4.0.5 + file-entry-cache@8.0.0: dependencies: flat-cache: 4.0.1 @@ -6254,36 +6687,69 @@ snapshots: lightningcss-android-arm64@1.32.0: optional: true + lightningcss-android-arm64@1.33.0: + optional: true + lightningcss-darwin-arm64@1.32.0: optional: true + lightningcss-darwin-arm64@1.33.0: + optional: true + lightningcss-darwin-x64@1.32.0: optional: true + lightningcss-darwin-x64@1.33.0: + optional: true + lightningcss-freebsd-x64@1.32.0: optional: true + lightningcss-freebsd-x64@1.33.0: + optional: true + lightningcss-linux-arm-gnueabihf@1.32.0: optional: true + lightningcss-linux-arm-gnueabihf@1.33.0: + optional: true + lightningcss-linux-arm64-gnu@1.32.0: optional: true + lightningcss-linux-arm64-gnu@1.33.0: + optional: true + lightningcss-linux-arm64-musl@1.32.0: optional: true + lightningcss-linux-arm64-musl@1.33.0: + optional: true + lightningcss-linux-x64-gnu@1.32.0: optional: true + lightningcss-linux-x64-gnu@1.33.0: + optional: true + lightningcss-linux-x64-musl@1.32.0: optional: true + lightningcss-linux-x64-musl@1.33.0: + optional: true + lightningcss-win32-arm64-msvc@1.32.0: optional: true + lightningcss-win32-arm64-msvc@1.33.0: + optional: true + lightningcss-win32-x64-msvc@1.32.0: optional: true + lightningcss-win32-x64-msvc@1.33.0: + optional: true + lightningcss@1.32.0: dependencies: detect-libc: 2.1.2 @@ -6300,6 +6766,22 @@ snapshots: lightningcss-win32-arm64-msvc: 1.32.0 lightningcss-win32-x64-msvc: 1.32.0 + lightningcss@1.33.0: + dependencies: + detect-libc: 2.1.2 + optionalDependencies: + lightningcss-android-arm64: 1.33.0 + lightningcss-darwin-arm64: 1.33.0 + lightningcss-darwin-x64: 1.33.0 + lightningcss-freebsd-x64: 1.33.0 + lightningcss-linux-arm-gnueabihf: 1.33.0 + lightningcss-linux-arm64-gnu: 1.33.0 + lightningcss-linux-arm64-musl: 1.33.0 + lightningcss-linux-x64-gnu: 1.33.0 + lightningcss-linux-x64-musl: 1.33.0 + lightningcss-win32-arm64-msvc: 1.33.0 + lightningcss-win32-x64-msvc: 1.33.0 + locate-path@6.0.0: dependencies: p-locate: 5.0.0 @@ -6405,6 +6887,8 @@ snapshots: nanoid@3.3.12: {} + nanoid@3.3.18: {} + napi-postinstall@0.3.4: {} natural-compare@1.4.0: {} @@ -6500,6 +6984,8 @@ snapshots: obliterator@1.6.1: {} + obug@2.1.4: {} + on-finished@2.4.1: dependencies: ee-first: 1.1.1 @@ -6573,6 +7059,8 @@ snapshots: picomatch@4.0.4: {} + picomatch@4.0.5: {} + possible-typed-array-names@1.1.0: {} postcss@8.4.31: @@ -6587,6 +7075,12 @@ snapshots: picocolors: 1.1.1 source-map-js: 1.2.1 + postcss@8.5.26: + dependencies: + nanoid: 3.3.18 + picocolors: 1.1.1 + source-map-js: 1.2.1 + prelude-ls@1.2.1: {} prop-types@15.8.1: @@ -6665,6 +7159,26 @@ snapshots: reusify@1.1.0: {} + rolldown@1.2.4: + dependencies: + '@oxc-project/types': 0.144.0 + '@rolldown/pluginutils': 1.0.1 + optionalDependencies: + '@rolldown/binding-android-arm64': 1.2.4 + '@rolldown/binding-darwin-arm64': 1.2.4 + '@rolldown/binding-darwin-x64': 1.2.4 + '@rolldown/binding-freebsd-x64': 1.2.4 + '@rolldown/binding-linux-arm-gnueabihf': 1.2.4 + '@rolldown/binding-linux-arm64-gnu': 1.2.4 + '@rolldown/binding-linux-arm64-musl': 1.2.4 + '@rolldown/binding-linux-ppc64-gnu': 1.2.4 + '@rolldown/binding-linux-s390x-gnu': 1.2.4 + '@rolldown/binding-linux-x64-gnu': 1.2.4 + '@rolldown/binding-linux-x64-musl': 1.2.4 + '@rolldown/binding-openharmony-arm64': 1.2.4 + '@rolldown/binding-win32-arm64-msvc': 1.2.4 + '@rolldown/binding-win32-x64-msvc': 1.2.4 + router@2.2.0: dependencies: debug: 4.4.3 @@ -6820,6 +7334,8 @@ snapshots: side-channel-map: 1.0.1 side-channel-weakmap: 1.0.2 + siginfo@2.0.0: {} + signal-exit@3.0.7: {} signal-exit@4.1.0: {} @@ -6835,8 +7351,12 @@ snapshots: stable-hash@0.0.5: {} + stackback@0.0.2: {} + statuses@2.0.2: {} + std-env@4.2.0: {} + stop-iteration-iterator@1.1.0: dependencies: es-errors: 1.3.0 @@ -6942,11 +7462,22 @@ snapshots: commander: 2.20.3 source-map-support: 0.5.21 + tinybench@2.9.0: {} + + tinyexec@1.3.0: {} + tinyglobby@0.2.16: dependencies: fdir: 6.5.0(picomatch@4.0.4) picomatch: 4.0.4 + tinyglobby@0.2.17: + dependencies: + fdir: 6.5.0(picomatch@4.0.5) + picomatch: 4.0.5 + + tinyrainbow@3.1.1: {} + to-regex-range@5.0.1: dependencies: is-number: 7.0.0 @@ -7089,6 +7620,49 @@ snapshots: vary@1.1.2: {} + vite@8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0): + dependencies: + lightningcss: 1.33.0 + picomatch: 4.0.5 + postcss: 8.5.26 + rolldown: 1.2.4 + tinyglobby: 0.2.17 + optionalDependencies: + '@types/node': 22.20.1 + esbuild: 0.27.3 + fsevents: 2.3.3 + jiti: 2.7.0 + terser: 5.16.9 + tsx: 4.22.0 + yaml: 2.9.0 + + vitest@4.1.10(@types/node@22.20.1)(vite@8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0)): + dependencies: + '@vitest/expect': 4.1.10 + '@vitest/mocker': 4.1.10(vite@8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0)) + '@vitest/pretty-format': 4.1.10 + '@vitest/runner': 4.1.10 + '@vitest/snapshot': 4.1.10 + '@vitest/spy': 4.1.10 + '@vitest/utils': 4.1.10 + es-module-lexer: 2.3.1 + expect-type: 1.4.0 + magic-string: 0.30.21 + obug: 2.1.4 + pathe: 2.0.3 + picomatch: 4.0.4 + std-env: 4.2.0 + tinybench: 2.9.0 + tinyexec: 1.3.0 + tinyglobby: 0.2.16 + tinyrainbow: 3.1.1 + vite: 8.2.1(@types/node@22.20.1)(esbuild@0.27.3)(jiti@2.7.0)(terser@5.16.9)(tsx@4.22.0)(yaml@2.9.0) + why-is-node-running: 2.3.0 + optionalDependencies: + '@types/node': 22.20.1 + transitivePeerDependencies: + - msw + web-streams-polyfill@4.0.0-beta.3: {} webidl-conversions@3.0.1: {} @@ -7147,6 +7721,11 @@ snapshots: dependencies: isexe: 3.1.5 + why-is-node-running@2.3.0: + dependencies: + siginfo: 2.0.0 + stackback: 0.0.2 + word-wrap@1.2.5: {} workerd@1.20260511.1: diff --git a/vitest.config.mts b/vitest.config.mts new file mode 100644 index 0000000..701b519 --- /dev/null +++ b/vitest.config.mts @@ -0,0 +1,18 @@ +import { defineConfig } from 'vitest/config' + +/** + * Two tests, not a suite — the daily counter and the crawl URL validator. + * Everything else in this repo is marketing pages where a type error and a + * design review already catch what matters; these two are the money and + * security paths. See docs/ai-chatbot-plan.md. + * + * Plain node environment: both units are pure logic, and the counter runs its + * real SQL against node:sqlite (D1 is SQLite). + */ +export default defineConfig({ + test: { + environment: 'node', + include: ['lib/**/*.test.ts'], + exclude: ['node_modules/**', '.next/**', '.open-next/**'], + }, +}) From 070a0ec41d58a155ca6d5bb2f3313b978055a283 Mon Sep 17 00:00:00 2001 From: harshit-epyc Date: Fri, 14 Aug 2026 18:36:28 +0530 Subject: [PATCH 02/12] docs: how to run the chatbot tool locally Secrets needed, migrations, where the verification code appears while email sending is stubbed, and how to preview the embed widget. Also records that OpenRouter's free quota is account-wide and needs $10 of credits to be usable, since that is what stops the tool working first. Co-Authored-By: Claude Opus 5 (1M context) --- docs/ai-chatbot-local-testing.md | 161 +++++++++++++++++++++++++++++++ 1 file changed, 161 insertions(+) create mode 100644 docs/ai-chatbot-local-testing.md diff --git a/docs/ai-chatbot-local-testing.md b/docs/ai-chatbot-local-testing.md new file mode 100644 index 0000000..f7b0b5f --- /dev/null +++ b/docs/ai-chatbot-local-testing.md @@ -0,0 +1,161 @@ +# AI Chatbot — Testing it locally + +Everything runs on your machine against a local database. Nothing here touches +staging or production. + +--- + +## 1. Secrets + +Create `.dev.vars` in the repo root. It is gitignored — never commit it. + +``` +OPENROUTER_API_KEY=sk-or-v1-… +TOOLS_IP_SALT=any-random-string-for-local +TOOLS_SESSIONS_PER_IP=100 +TOOLS_EMBED_ALLOW_ANY_ORIGIN=true +``` + +| Key | Why | +|---|---| +| `OPENROUTER_API_KEY` | The only one you cannot invent. Get it from openrouter.ai | +| `TOOLS_IP_SALT` | Hashes visitor IPs and verification codes. Any string locally | +| `TOOLS_SESSIONS_PER_IP` | Local only. Without it you get 3 crawls a day, because localhost has no per-visitor IP and everything shares one counter | +| `TOOLS_EMBED_ALLOW_ANY_ORIGIN` | Local only. An embed key is bound to the site it was minted for, so a widget can never be previewed on localhost without it | + +**The last two must stay unset in staging and production.** They switch off real +limits. + +> **⚠️ OpenRouter needs $10 of credits.** Without them the account allows about +> 50 free requests a day *across all models* — the free tiers share one quota, +> so the fallback chain does not help. With credits it is 1,000 a day and still +> costs nothing per message. Without this you will hit `429` within minutes. + +--- + +## 2. Database + +Apply the three migrations to the local database once: + +```bash +pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --local \ + --file=db/migrations/0003_tool_sessions.sql +pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --local \ + --file=db/migrations/0004_tool_embeds.sql +pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --local \ + --file=db/migrations/0005_tool_verifications.sql +``` + +--- + +## 3. Run it + +```bash +pnpm dev +``` + +`.dev.vars` is read at startup, so **restart after changing it**. Look for +`Using secrets defined in .dev.vars` in the output. + +--- + +## 4. The tool + +Open **http://localhost:3000/tools/ai-chatbot** + +1. Paste any real website address, press **Read my site** +2. Watch the pages arrive one by one +3. Ask it questions — try one the site cannot answer, like *"what does it cost?"* +4. Click **Skip to my report**, or use all 8 questions + +The report's three measured scores appear immediately. The headline +"X of 10" is scored in the background and lands within about 20 seconds. + +**Sites worth trying:** `linear.app` and `stripe.com` score well; most small +agency sites score 3–5; a React SPA shows the "we could not read your site" +path. + +--- + +## 5. The embed — where to get the code + +At the bottom of the report: + +1. Enter any email → **Send me a code** +2. **The code is printed in your `pnpm dev` terminal**, in a box: + + ``` + ─────────── EMAIL (not sent — no provider configured) ─────────── + To: you@company.com + Subject: 059052 is your EPYC verification code + ``` + + No email is actually sent — there is no provider yet. The UI says so. + +3. Type the 6 digits → **Verify** +4. The embed snippet appears with a copy button + +Limits while testing: 3 codes per email per day, 3 per session, 10 minute +expiry, 5 wrong attempts and the code dies. + +--- + +## 6. Seeing the widget + +Create `public/embed-test.html` (gitignored) and paste your snippet into it: + +```html + + + +

Pretend customer site

+ + + +``` + +Open **http://localhost:3000/embed-test.html** and hard-refresh. + +A circular button appears bottom-right. Open it and ask something about the +site you crawled — it answers from those pages only. + +**The `src` must point at `http://localhost:3000`.** If the snippet says +`https://epyc.in`, change it — the live site does not have this code yet. + +This only works locally because of `TOOLS_EMBED_ALLOW_ANY_ORIGIN`. In +production the key is bound to the crawled domain and this page would get a +`403`, which is the binding working correctly. + +--- + +## Resetting when you hit a limit + +```bash +# all daily limits: crawls, messages, verification codes +pnpm exec wrangler d1 execute epyc-contact-form-aug-26-live-staging --local \ + --command "DELETE FROM tool_counters" +``` + +Other useful queries: + +```sql +-- what has been crawled +SELECT host, status, pages_crawled, messages_used FROM tool_sessions +ORDER BY created_at DESC LIMIT 5; + +-- issued embed keys +SELECT key, bound_host, email FROM tool_embeds ORDER BY created_at DESC; +``` + +--- + +## When something does not work + +| Symptom | Cause | +|---|---| +| "The assistant is unavailable" | `OPENROUTER_API_KEY` missing, or dev server not restarted after adding it | +| `429` from the model | OpenRouter free quota. Add $10 of credits | +| "You've used your 3 checks" | `TOOLS_SESSIONS_PER_IP` not set, or not restarted. Or clear `tool_counters` | +| Widget does not appear | Snippet points at the wrong origin, or dev server predates `TOOLS_EMBED_ALLOW_ANY_ORIGIN`. Check the browser console for a `403` | +| Same site returns instantly | Working as intended — crawls are reused for 24h. Use **Read my site again** to force a fresh one | +| Report says it could not be built | Old session from before a fix. Crawl again | From ca26cc7bb76c41372506298836f6a9d9df2450b9 Mon Sep 17 00:00:00 2001 From: harshit-epyc Date: Mon, 17 Aug 2026 16:02:19 +0530 Subject: [PATCH 03/12] feat(embed): host-safe origin binding, split dev/live quotas, owner recrawl MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Origin binding no longer strips labels. Deriving an apex from a hostname needs the Public Suffix List, and without it `random.vercel.app` reduces to `vercel.app`, which would let one key answer for every site Vercel hosts. The same holds for `*.myshopify.com`, `*.github.io` and `*.framer.website`. The bound host is now stored as crawled and compared as stored: exact host, `www`, and localhost. The subdomain wildcard is gone for the same reason. Localhost is accepted for any key, so a customer's developer can test before installing, and those messages are counted separately (`embed-dev:`, 20 a day) from live traffic (`embed:`, 50 a day). Two consequences: a laptop can never spend a live site's allowance, and a domain that crawls three times and claims three keys still gets 50 messages rather than 150. This replaces `TOOLS_EMBED_ALLOW_ANY_ORIGIN`, which was a global switch we could never hand to a customer. A rejected origin now names the bound host, because "not available" reads as broken to whoever is installing it. No per-IP cap on embed messages. One IP sending 50 messages is either the owner testing after install or someone draining the quota, and there is no way to tell them apart. Capping it punishes the first; the second costs nothing on a free model. Manual recrawl ships as a signed link rather than a button. The widget is shown to the customer's own visitors, and a recrawl points twenty requests at their server, so it cannot be something a stranger can press. The token is an HMAC of the key, so there is no table, no expiry to sweep, and revoking the embed kills the link. Three a day per domain. The embed's session id never changes, so the snippet already pasted into their HTML keeps working — only the pages underneath are swapped, and a recrawl that comes back empty is rejected rather than blanking a live bot. The link is shown on screen at claim time, not emailed: there is still no email provider configured, so a link we only sent by email would reach nobody. Also: widget history moves to sessionStorage, keyed per embed. It was held in a plain variable, so the conversation reset every time a visitor clicked a link. 11 tests on the origin check, mostly the hosting-suffix cases. Co-Authored-By: Claude Opus 5 (1M context) --- app/(my-app)/tools/ai-chatbot/manage/page.tsx | 37 +++++ app/api/embed/chatbot.js/route.ts | 25 +++- app/api/embed/chatbot/message/route.ts | 18 ++- app/api/tools/chatbot/embed/route.ts | 11 ++ app/api/tools/chatbot/recrawl/route.ts | 116 +++++++++++++++ cloudflare-env.secrets.d.ts | 7 - components/sections/chatbot-tool.tsx | 24 ++- components/sections/embed-manage.tsx | 109 ++++++++++++++ docs/ai-chatbot-local-testing.md | 43 ++++-- lib/tools/chatbot/embed.test.ts | 83 +++++++++++ lib/tools/chatbot/embed.ts | 137 ++++++++++++++---- lib/tools/chatbot/schema.ts | 6 + lib/tools/session.ts | 38 ++++- 13 files changed, 591 insertions(+), 63 deletions(-) create mode 100644 app/(my-app)/tools/ai-chatbot/manage/page.tsx create mode 100644 app/api/tools/chatbot/recrawl/route.ts create mode 100644 components/sections/embed-manage.tsx create mode 100644 lib/tools/chatbot/embed.test.ts diff --git a/app/(my-app)/tools/ai-chatbot/manage/page.tsx b/app/(my-app)/tools/ai-chatbot/manage/page.tsx new file mode 100644 index 0000000..824aa05 --- /dev/null +++ b/app/(my-app)/tools/ai-chatbot/manage/page.tsx @@ -0,0 +1,37 @@ +import type { Metadata } from 'next' +import { SiteNav } from '@/components/site-nav' +import { EmbedManage } from '@/components/sections/embed-manage' +import { Container } from '@/components/ui/container' +import { Section } from '@/components/ui/section' + +/** + * The manage page for a claimed embed, reached from the signed link handed to + * the owner when they claimed it. One button: read my site again. + * + * Never indexed and never linked from the site. The link itself is the + * credential — see lib/tools/chatbot/embed.ts → manageToken. + */ +export const metadata: Metadata = { + title: 'Refresh your assistant', + robots: { index: false, follow: false }, +} + +export default async function ManageEmbedPage({ + searchParams, +}: { + searchParams: Promise<{ key?: string; t?: string }> +}) { + const { key = '', t = '' } = await searchParams + + return ( + <> +
+ + + +
+ + + + ) +} diff --git a/app/api/embed/chatbot.js/route.ts b/app/api/embed/chatbot.js/route.ts index be0b51a..d8719ac 100644 --- a/app/api/embed/chatbot.js/route.ts +++ b/app/api/embed/chatbot.js/route.ts @@ -100,7 +100,21 @@ const WIDGET = String.raw` return b; } + // History survives navigation. Without this the conversation resets every + // time the visitor clicks a link, because the script reloads with the page. + // sessionStorage, not localStorage: it dies with the tab, which keeps this + // out of cookie-consent territory. Scoped per key so two bots cannot mix. + var STORE = 'epyc_bot_' + key; var history = []; + try { + history = JSON.parse(sessionStorage.getItem(STORE) || '[]'); + if (!Array.isArray(history)) history = []; + } catch (e) { history = []; } + + function remember() { + try { sessionStorage.setItem(STORE, JSON.stringify(history.slice(-8))); } catch (e) {} + } + var busy = false; function ask(text) { @@ -132,6 +146,7 @@ const WIDGET = String.raw` if (chunk.done) { history.push({ role: 'user', content: text }); history.push({ role: 'assistant', content: answer }); + remember(); busy = false; return; } @@ -180,7 +195,15 @@ const WIDGET = String.raw` input.focus(); if (!mounted) { mounted = true; - bubble('bot', 'Hi — ask me anything about this site.'); + // Replay what was said before the visitor changed page. Skipping this + // would leave the panel blank while the model still had the context. + if (history.length) { + history.forEach(function (turn) { + bubble(turn.role === 'user' ? 'you' : 'bot', turn.content); + }); + } else { + bubble('bot', 'Hi — ask me anything about this site.'); + } } } }); diff --git a/app/api/embed/chatbot/message/route.ts b/app/api/embed/chatbot/message/route.ts index e3802d1..c1f34d9 100644 --- a/app/api/embed/chatbot/message/route.ts +++ b/app/api/embed/chatbot/message/route.ts @@ -21,8 +21,9 @@ import { loadPages } from '@/lib/tools/session' * means something. An iframe we host would report our own origin instead. * * Binding is not airtight — a non-browser client can send any Origin it likes. - * The per-key daily cap is what bounds that: the worst case is a fixed amount - * of free-tier inference, not an open tap. + * The per-host daily cap is what bounds that: the worst case is a fixed amount + * of free-tier inference, not an open tap. Localhost is allowed for any key and + * counted separately, so testing cannot spend a live site's allowance. */ export async function OPTIONS(req: Request) { @@ -51,12 +52,13 @@ export async function POST(req: Request) { return NextResponse.json({ ok: false, error: 'This assistant is not available.' }, { status: 404 }) } - // The binding. A key minted for acme.com answers only for acme.com. - if (!originAllowed(origin, embed.bound_host, { - allowAny: env.TOOLS_EMBED_ALLOW_ANY_ORIGIN === 'true', - })) { + // The binding. A key minted for acme.com answers only for acme.com — plus + // localhost, so the customer's developer can test before installing. + if (!originAllowed(origin, embed.bound_host)) { + // Name the host. A developer seeing this on staging.acme.com needs to know + // why, not guess — "not available" reads as broken. return NextResponse.json( - { ok: false, error: 'This assistant is not available on this domain.' }, + { ok: false, error: `This assistant only runs on ${embed.bound_host}.` }, { status: 403 }, ) } @@ -72,7 +74,7 @@ export async function POST(req: Request) { ) } - if (!(await consumeEmbedMessage(db, embed.key))) { + if (!(await consumeEmbedMessage(db, embed, origin))) { return NextResponse.json( { ok: false, error: 'This assistant has reached its limit for today.' }, { status: 429, headers: cors }, diff --git a/app/api/tools/chatbot/embed/route.ts b/app/api/tools/chatbot/embed/route.ts index 472dd64..e95496a 100644 --- a/app/api/tools/chatbot/embed/route.ts +++ b/app/api/tools/chatbot/embed/route.ts @@ -5,6 +5,8 @@ import { createEmbed, embedSnippet, findEmbedBySession, + manageToken, + manageUrl, mintKey, } from '@/lib/tools/chatbot/embed' import { isVerified } from '@/lib/tools/chatbot/verification' @@ -58,6 +60,13 @@ export async function POST(req: Request) { } const origin = new URL(req.url).origin + const salt = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + // The manage link is returned with the snippet, and shown on screen rather + // than emailed: there is no email provider configured yet, so a link we only + // sent by email would reach nobody. It is derived from the key, so it can be + // handed out again on a re-claim without storing anything. + const manage = async (key: string) => manageUrl(key, await manageToken(key, salt), origin) // Claiming twice returns the same key rather than minting a second one — // otherwise a refresh silently orphans the snippet they already pasted. @@ -68,6 +77,7 @@ export async function POST(req: Request) { key: existing.key, host: existing.bound_host, snippet: embedSnippet(existing.key, origin), + manageUrl: await manage(existing.key), alreadyClaimed: true, }) } @@ -92,6 +102,7 @@ export async function POST(req: Request) { key, host: session.host, snippet: embedSnippet(key, origin), + manageUrl: await manage(key), alreadyClaimed: false, }) } diff --git a/app/api/tools/chatbot/recrawl/route.ts b/app/api/tools/chatbot/recrawl/route.ts new file mode 100644 index 0000000..50579f1 --- /dev/null +++ b/app/api/tools/chatbot/recrawl/route.ts @@ -0,0 +1,116 @@ +import { NextResponse } from 'next/server' +import { getCloudflareContext } from '@opennextjs/cloudflare' +import { crawlSite } from '@/lib/crawl/fetch-pages' +import { validateUrl } from '@/lib/crawl/validate-url' +import { recrawlSchema } from '@/lib/tools/chatbot/schema' +import { + EMBED_RECRAWLS_PER_DAY, + consumeRecrawl, + findEmbedByKey, + isEmbedKey, + manageToken, +} from '@/lib/tools/chatbot/embed' +import { scoreDeterministic } from '@/lib/tools/chatbot/diagnosis' +import { clearPages, finishSession, getSession, savePages, saveDiagnosis } from '@/lib/tools/session' + +/** + * Re-read a live embed's site and replace its corpus. + * + * Reached only from the signed manage link — there is deliberately no button in + * the widget, because the widget is shown to the customer's visitors and a + * recrawl points 20 pages of traffic at their own server. The owner holds the + * link; nobody else has a way in. + * + * The embed's `session_id` never changes, so the snippet already pasted into + * their HTML keeps working. Only the pages underneath it are swapped. + * + * Plain JSON, not SSE: one person clicking one button can wait with a spinner. + * The live progress stream on the demo crawl exists to make a stranger trust + * the tool, which is not this audience. + */ +export async function POST(req: Request) { + const json = await req.json().catch(() => null) + const parsed = recrawlSchema.safeParse(json) + if (!parsed.success || !isEmbedKey(parsed.data.key)) { + return NextResponse.json({ ok: false, error: 'Invalid request.' }, { status: 400 }) + } + + const { env } = getCloudflareContext() + const db = env.DB + const salt = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + const embed = await findEmbedByKey(db, parsed.data.key) + if (!embed || embed.status !== 'active') { + return NextResponse.json({ ok: false, error: 'That link is no longer valid.' }, { status: 404 }) + } + + // ponytail: plain string compare. This is an HMAC over a network round trip, + // not a local secret check — a timing oracle here is not a practical attack. + if (parsed.data.t !== (await manageToken(embed.key, salt))) { + return NextResponse.json({ ok: false, error: 'That link is no longer valid.' }, { status: 403 }) + } + + if (!(await consumeRecrawl(db, embed.bound_host))) { + return NextResponse.json( + { + ok: false, + error: `You've refreshed ${embed.bound_host} ${EMBED_RECRAWLS_PER_DAY} times today. It resets at midnight UTC.`, + }, + { status: 429 }, + ) + } + + const session = await getSession(db, embed.session_id) + if (!session) { + return NextResponse.json({ ok: false, error: 'That link is no longer valid.' }, { status: 404 }) + } + + // Re-validated, not trusted from the row. The stored URL was safe when it was + // crawled; this is the same gate the demo crawl runs, applied again. + const checked = validateUrl(session.target_url) + if (!checked.ok) { + return NextResponse.json({ ok: false, error: checked.reason }, { status: 400 }) + } + + try { + const result = await crawlSite(checked.url) + const readable = result.pages.filter((p) => !p.isEmpty) + + // A failed or empty recrawl must never wipe a working bot. Their site may + // simply have been down for the twenty seconds we were reading it, and the + // old corpus is better than none on a page their customers are using. + if (readable.length === 0) { + return NextResponse.json( + { + ok: false, + error: 'We could not read your site just now, so your bot is unchanged. Try again shortly.', + }, + { status: 502 }, + ) + } + + await clearPages(db, embed.session_id) + await savePages(db, embed.session_id, result.pages) + await finishSession(db, embed.session_id, 'ready', result.pages.length) + await saveDiagnosis(db, embed.session_id, scoreDeterministic(result.pages, result.signals)) + + const crawledAt = new Date().toISOString() + await db + .prepare('UPDATE tool_embeds SET crawled_at = ? WHERE key = ?') + .bind(crawledAt, embed.key) + .run() + + return NextResponse.json({ + ok: true, + host: embed.bound_host, + pages: readable.length, + crawledAt, + }) + } catch (err) { + console.error('recrawl failed', err) + return NextResponse.json( + { ok: false, error: 'We could not read your site just now, so your bot is unchanged.' }, + { status: 502 }, + ) + } +} diff --git a/cloudflare-env.secrets.d.ts b/cloudflare-env.secrets.d.ts index 77af03e..d173fd4 100644 --- a/cloudflare-env.secrets.d.ts +++ b/cloudflare-env.secrets.d.ts @@ -27,13 +27,6 @@ declare namespace Cloudflare { * and three crawls exhausts the day. Leave unset in staging and production. */ TOOLS_SESSIONS_PER_IP?: string - /** - * Lets an embed answer from any Origin. Local development only — a key is - * bound to the site it was minted for, so a widget cannot otherwise be - * previewed from localhost. Leave unset in staging and production: without - * the binding, a public key works from anywhere. - */ - TOOLS_EMBED_ALLOW_ANY_ORIGIN?: string /** * Transactional email provider key. Unset today, which makes * lib/tools/email.ts log verification codes instead of sending them. diff --git a/components/sections/chatbot-tool.tsx b/components/sections/chatbot-tool.tsx index f2b6872..412b23a 100644 --- a/components/sections/chatbot-tool.tsx +++ b/components/sections/chatbot-tool.tsx @@ -755,6 +755,7 @@ function ClaimEmbed({ sessionId, host }: { sessionId: string; host: string }) { const [email, setEmail] = useState('') const [code, setCode] = useState('') const [snippet, setSnippet] = useState(null) + const [manageUrl, setManageUrl] = useState(null) const [error, setError] = useState(null) const [notice, setNotice] = useState(null) const [busy, setBusy] = useState(false) @@ -814,9 +815,15 @@ function ClaimEmbed({ sessionId, host }: { sessionId: string; host: string }) { headers: { 'content-type': 'application/json' }, body: JSON.stringify({ sessionId, email: email.trim() }), }) - const body = (await res.json()) as { ok: boolean; snippet?: string; error?: string } + const body = (await res.json()) as { + ok: boolean + snippet?: string + manageUrl?: string + error?: string + } if (body.ok && body.snippet) { setSnippet(body.snippet) + setManageUrl(body.manageUrl ?? null) setStep('done') setNotice(null) } else { @@ -868,6 +875,21 @@ function ClaimEmbed({ sessionId, host }: { sessionId: string; host: string }) { Works on {host} only. Free, and it answers from the pages we just read. + + {manageUrl && ( +
+

+ Keep this link. It is how you refresh the bot when your site changes, and it is + the only copy — we cannot send it to you yet. +

+ + {manageUrl} + +
+ )} ) : step === 'code' ? ( <> diff --git a/components/sections/embed-manage.tsx b/components/sections/embed-manage.tsx new file mode 100644 index 0000000..2da66cd --- /dev/null +++ b/components/sections/embed-manage.tsx @@ -0,0 +1,109 @@ +'use client' + +import { useState } from 'react' +import { Button } from '@/components/ui/button' +import { Container } from '@/components/ui/container' +import { Section } from '@/components/ui/section' +import { SectionHeading } from '@/components/ui/section-heading' + +/** + * The owner's one-button page for a live embed: read my site again. + * + * Deliberately not in the widget. The widget is shown to the customer's own + * visitors, and a recrawl points twenty page requests at their server — that + * cannot be a button a stranger can press. The signed link in `?key` + `?t` is + * the only way here, and the route re-checks both. + * + * No status is fetched on load: the page holds nothing worth a second route + * until the button is pressed, and the response carries everything worth + * showing. + */ +export function EmbedManage({ embedKey, token }: { embedKey: string; token: string }) { + const [busy, setBusy] = useState(false) + const [error, setError] = useState(null) + const [done, setDone] = useState<{ host: string; pages: number } | null>(null) + + const linkLooksValid = embedKey.startsWith('ek_live_') && token.length > 0 + + async function recrawl() { + if (busy) return + setBusy(true) + setError(null) + setDone(null) + + try { + const res = await fetch('/api/tools/chatbot/recrawl', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ key: embedKey, t: token }), + }) + const body = (await res.json()) as { + ok: boolean + host?: string + pages?: number + error?: string + } + + if (body.ok && body.host) { + setDone({ host: body.host, pages: body.pages ?? 0 }) + } else { + setError(body.error ?? 'Something went wrong. Try again.') + } + } catch { + setError('Something went wrong. Try again.') + } finally { + setBusy(false) + } + } + + return ( +
+ +
+ + Refresh your assistant + + + {!linkLooksValid ? ( +

+ This link is incomplete. Use the full link from the page where you got your embed + code. +

+ ) : done ? ( + <> +

+ Read {done.pages} {done.pages === 1 ? 'page' : 'pages'} of {done.host}. Your + assistant now answers from what is on your site today. Nothing on your site needs + changing — the code you pasted is unchanged. +

+ + + ) : ( + <> +

+ Your assistant answers from a snapshot of your site. Changed something? Read it + again and the assistant picks it up. Takes about 20 seconds, and the code on your + site stays exactly as it is. +

+ +
+ + Three times a day. +
+ + )} + + {error && ( +

+ {error} +

+ )} +
+
+
+ ) +} diff --git a/docs/ai-chatbot-local-testing.md b/docs/ai-chatbot-local-testing.md index f7b0b5f..49b1a6f 100644 --- a/docs/ai-chatbot-local-testing.md +++ b/docs/ai-chatbot-local-testing.md @@ -13,18 +13,20 @@ Create `.dev.vars` in the repo root. It is gitignored — never commit it. OPENROUTER_API_KEY=sk-or-v1-… TOOLS_IP_SALT=any-random-string-for-local TOOLS_SESSIONS_PER_IP=100 -TOOLS_EMBED_ALLOW_ANY_ORIGIN=true ``` | Key | Why | |---|---| | `OPENROUTER_API_KEY` | The only one you cannot invent. Get it from openrouter.ai | -| `TOOLS_IP_SALT` | Hashes visitor IPs and verification codes. Any string locally | +| `TOOLS_IP_SALT` | Hashes visitor IPs, verification codes, and embed manage links. Any string locally | | `TOOLS_SESSIONS_PER_IP` | Local only. Without it you get 3 crawls a day, because localhost has no per-visitor IP and everything shares one counter | -| `TOOLS_EMBED_ALLOW_ANY_ORIGIN` | Local only. An embed key is bound to the site it was minted for, so a widget can never be previewed on localhost without it | -**The last two must stay unset in staging and production.** They switch off real -limits. +**`TOOLS_SESSIONS_PER_IP` must stay unset in staging and production.** It +switches off a real limit. + +Nothing is needed to preview the widget. An embed key answers any `localhost` +origin by design, on a separate 20-a-day allowance that cannot touch the live +site's 50. > **⚠️ OpenRouter needs $10 of credits.** Without them the account allows about > 50 free requests a day *across all models* — the free tiers share one quota, @@ -93,7 +95,10 @@ At the bottom of the report: No email is actually sent — there is no provider yet. The UI says so. 3. Type the 6 digits → **Verify** -4. The embed snippet appears with a copy button +4. The embed snippet appears with a copy button, and under it a **manage link**. + That link is the only way to refresh the bot later — there is no button for + it in the widget, because the widget is shown to the customer's visitors. + Copy it somewhere; nothing emails it to you. Limits while testing: 3 codes per email per day, 3 per session, 10 minute expiry, 5 wrong attempts and the code dies. @@ -122,9 +127,26 @@ site you crawled — it answers from those pages only. **The `src` must point at `http://localhost:3000`.** If the snippet says `https://epyc.in`, change it — the live site does not have this code yet. -This only works locally because of `TOOLS_EMBED_ALLOW_ANY_ORIGIN`. In -production the key is bound to the crawled domain and this page would get a -`403`, which is the binding working correctly. +Localhost is allowed for any key on purpose, so a customer's developer can test +before installing. Those messages come out of a separate 20-a-day bucket +(`embed-dev:`), so testing can never spend the live site's 50. Any other +domain gets a `403` naming the bound host, which is the binding working. + +The conversation survives page navigation — it is held in `sessionStorage` per +key, so add a second page to `embed-test.html` and the thread continues. + +--- + +## 7. Refreshing a bot + +Open the manage link from step 5. One button: **Read my site again**. It +re-reads the site, swaps the pages under the same key, and leaves the snippet on +the customer's site untouched. + +Capped at 3 a day per domain. Clear `tool_counters` to reset while testing. + +A recrawl that comes back empty is rejected and the old pages are kept — a site +that is down for twenty seconds must not blank a live bot. --- @@ -156,6 +178,7 @@ SELECT key, bound_host, email FROM tool_embeds ORDER BY created_at DESC; | "The assistant is unavailable" | `OPENROUTER_API_KEY` missing, or dev server not restarted after adding it | | `429` from the model | OpenRouter free quota. Add $10 of credits | | "You've used your 3 checks" | `TOOLS_SESSIONS_PER_IP` not set, or not restarted. Or clear `tool_counters` | -| Widget does not appear | Snippet points at the wrong origin, or dev server predates `TOOLS_EMBED_ALLOW_ANY_ORIGIN`. Check the browser console for a `403` | +| Widget does not appear | Snippet points at the wrong origin. Check the browser console for a `403` — the message names the host the key is bound to | +| "That link is no longer valid" on manage | `TOOLS_IP_SALT` changed since the key was claimed. The token is derived from it. Claim again | | Same site returns instantly | Working as intended — crawls are reused for 24h. Use **Read my site again** to force a fresh one | | Report says it could not be built | Old session from before a fix. Crawl again | diff --git a/lib/tools/chatbot/embed.test.ts b/lib/tools/chatbot/embed.test.ts new file mode 100644 index 0000000..c9d4c0b --- /dev/null +++ b/lib/tools/chatbot/embed.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from 'vitest' +import { isEmbedKey, isLocalOrigin, originAllowed } from './embed' + +/** + * The origin check is what stops a public key answering on someone else's site. + * It is not a hard security boundary — `Origin` is forgeable from any non-browser + * client, which is what the daily cap is for — but it is the only thing standing + * between a copied key and a working bot, so its edges are worth pinning down. + * + * The cases that matter most are the hosting suffixes: label stripping here + * would hand one key authority over every site on `vercel.app`. + */ + +describe('originAllowed', () => { + it('accepts the exact bound host and its www form', () => { + expect(originAllowed('https://acme.com', 'acme.com')).toBe(true) + expect(originAllowed('https://www.acme.com', 'acme.com')).toBe(true) + expect(originAllowed('https://acme.com', 'www.acme.com')).toBe(true) + }) + + it('ignores the port and is case insensitive', () => { + expect(originAllowed('https://ACME.com:443', 'acme.com')).toBe(true) + }) + + it('rejects a different site', () => { + expect(originAllowed('https://evil.com', 'acme.com')).toBe(false) + expect(originAllowed('https://acme.com.evil.com', 'acme.com')).toBe(false) + expect(originAllowed('https://notacme.com', 'acme.com')).toBe(false) + }) + + it('rejects subdomains of the bound host', () => { + // Deliberate: allowing these means allowing them for a bound host that is + // itself a hosting suffix. See the sibling case below. + expect(originAllowed('https://staging.acme.com', 'acme.com')).toBe(false) + }) + + it('never lets one tenant of a hosting suffix answer for another', () => { + expect(originAllowed('https://someone-else.vercel.app', 'mine.vercel.app')).toBe(false) + expect(originAllowed('https://vercel.app', 'mine.vercel.app')).toBe(false) + expect(originAllowed('https://other.myshopify.com', 'shop.myshopify.com')).toBe(false) + expect(originAllowed('https://victim.github.io', 'github.io')).toBe(false) + }) + + it('accepts the bound tenant of a hosting suffix', () => { + expect(originAllowed('https://mine.vercel.app', 'mine.vercel.app')).toBe(true) + }) + + it('rejects a missing or unusable Origin', () => { + expect(originAllowed(null, 'acme.com')).toBe(false) + expect(originAllowed('', 'acme.com')).toBe(false) + expect(originAllowed('null', 'acme.com')).toBe(false) + expect(originAllowed('file:///tmp/x.html', 'acme.com')).toBe(false) + expect(originAllowed('javascript:alert(1)', 'acme.com')).toBe(false) + }) + + it('accepts localhost on any port, for any key', () => { + expect(originAllowed('http://localhost:3000', 'acme.com')).toBe(true) + expect(originAllowed('http://127.0.0.1:8788', 'acme.com')).toBe(true) + expect(originAllowed('http://[::1]:3000', 'acme.com')).toBe(true) + }) + + it('does not accept a hostname that merely contains localhost', () => { + expect(originAllowed('https://localhost.evil.com', 'acme.com')).toBe(false) + expect(originAllowed('https://notlocalhost', 'acme.com')).toBe(false) + }) +}) + +describe('isLocalOrigin', () => { + it('separates dev traffic from live traffic, so the two caps cannot blur', () => { + expect(isLocalOrigin('http://localhost:5173')).toBe(true) + expect(isLocalOrigin('https://acme.com')).toBe(false) + expect(isLocalOrigin(null)).toBe(false) + }) +}) + +describe('isEmbedKey', () => { + it('accepts a minted key and rejects near misses', () => { + expect(isEmbedKey('ek_live_' + 'a'.repeat(32))).toBe(true) + expect(isEmbedKey('ek_live_' + 'a'.repeat(31))).toBe(false) + expect(isEmbedKey('ek_test_' + 'a'.repeat(32))).toBe(false) + expect(isEmbedKey('ek_live_' + 'Z'.repeat(32))).toBe(false) + }) +}) diff --git a/lib/tools/chatbot/embed.ts b/lib/tools/chatbot/embed.ts index e9fb57d..113bccc 100644 --- a/lib/tools/chatbot/embed.ts +++ b/lib/tools/chatbot/embed.ts @@ -6,13 +6,14 @@ * * 1. A key minted for acme.com only answers requests whose Origin is * acme.com or www.acme.com. CORS is set to that host, never `*`. - * 2. Each key carries its own daily message counter. Origin can be forged by + * 2. The bound host carries a daily message counter. Origin can be forged by * a non-browser client, so the cap is what bounds the worst case — a * bounded amount of free inference, not an open tap. * 3. Keys can be revoked with one row update. */ import { bumpCounter } from '../counters' +import { hmacHex } from '../session' export type EmbedRow = { key: string @@ -23,9 +24,27 @@ export type EmbedRow = { crawled_at: string } -/** Messages per embed per day. A runaway backstop, not a product limit. */ +/** + * Messages per bound host per day. A runaway backstop, not a product limit. + * + * Keyed on the host rather than the key: one domain can crawl three times in a + * day and claim three sessions, which would otherwise be 150 messages. + */ export const EMBED_DAILY_MESSAGES = 50 +/** + * Messages from the customer's own machine per day. + * + * A separate bucket, so a laptop can never drain the live site's allowance. + * Any localhost origin is accepted for any key — the key is public and sits in + * their page source, and forging `Origin` from curl was always possible anyway, + * so refusing localhost bought nothing and blocked every developer. + */ +export const EMBED_DEV_DAILY_MESSAGES = 20 + +/** Recrawls per bound host per day, from the signed manage link. */ +export const EMBED_RECRAWLS_PER_DAY = 3 + const KEY_PREFIX = 'ek_live_' /** 128 bits of randomness, hex encoded. Not a secret, but must be unguessable. */ @@ -38,36 +57,59 @@ export function isEmbedKey(value: string): boolean { return /^ek_live_[0-9a-f]{32}$/.test(value) } +/** The hostname of an Origin header, lowercased, without the port. */ +function hostnameOf(origin: string | null): string | null { + if (!origin) return null + try { + const url = new URL(origin) + if (url.protocol !== 'https:' && url.protocol !== 'http:') return null + return url.hostname.toLowerCase() + } catch { + return null + } +} + /** - * Does this Origin belong to the host the key was minted for? + * Is this request coming from the customer's own machine? * - * Apex and `www` only. Subdomains are deliberately not accepted: a key issued - * for acme.com should not answer for anything.acme.com, because we never - * crawled those and the bot would confidently answer from the wrong corpus. + * One definition, used by both the origin check and the daily cap, so the two + * can never disagree about what counts as local. Any port matches — `hostname` + * drops it. */ -export function originAllowed( - origin: string | null, - boundHost: string, - opts: { allowAny?: boolean } = {}, -): boolean { - if (!origin) return false +export function isLocalOrigin(origin: string | null): boolean { + const host = hostnameOf(origin) + return host === 'localhost' || host === '127.0.0.1' || host === '[::1]' || host === '::1' +} - // Local development only. A key is bound to the crawled site, so a widget - // can never be previewed from localhost without this. Gated on an env var - // that is unset in staging and production — see cloudflare-env.secrets.d.ts. - if (opts.allowAny) return true +/** + * Does this Origin belong to the host the key was minted for? + * + * Exactly the crawled host, plus `www`, plus localhost. Three rules, and the + * important one is what is missing: **no label stripping, ever.** + * + * Deriving an apex from a hostname needs the Public Suffix List. Without it, + * `random.vercel.app` strips to `vercel.app` and a single key would answer for + * every site Vercel hosts. The same holds for `*.myshopify.com`, + * `*.github.io`, `*.framer.website`. So `bound_host` is stored as crawled and + * compared as stored. + * + * A wildcard for subdomains was considered and rejected for the same reason: + * crawling a bare public suffix would hand out a key covering everyone on it. + * + * ponytail: no per-key origin allowlist. A customer who needs the widget on + * `staging.acme.com` gets a 403 naming the bound host and installs on + * production instead; testing happens on localhost, which is allowed. Upgrade + * path is an `extra_origins` column on tool_embeds, set explicitly from the + * manage page — never inferred from the string. + */ +export function originAllowed(origin: string | null, boundHost: string): boolean { + if (isLocalOrigin(origin)) return true - let host: string - try { - const url = new URL(origin) - if (url.protocol !== 'https:' && url.protocol !== 'http:') return false - host = url.host.toLowerCase() - } catch { - return false - } + const host = hostnameOf(origin) + if (!host) return false - const apex = boundHost.toLowerCase().replace(/^www\./, '') - return host === apex || host === `www.${apex}` + const bound = boundHost.toLowerCase().replace(/^www\./, '') + return host === bound || host === `www.${bound}` } /** CORS headers for a bound embed. Never `*` — the allowlist is one host. */ @@ -118,9 +160,26 @@ export async function findEmbedByKey(db: D1Database, key: string): Promise() } -/** Consume one of this key's daily messages. False means the cap is reached. */ -export async function consumeEmbedMessage(db: D1Database, key: string): Promise { - return bumpCounter(db, `embed:${key}`, EMBED_DAILY_MESSAGES) +/** + * Consume one daily message. False means the cap is reached. + * + * Two buckets. Production traffic is capped per bound host; the customer's own + * machine gets a smaller separate allowance, so testing can never take the live + * site's messages away from its real visitors. + */ +export async function consumeEmbedMessage( + db: D1Database, + embed: EmbedRow, + origin: string | null, +): Promise { + return isLocalOrigin(origin) + ? bumpCounter(db, `embed-dev:${embed.key}`, EMBED_DEV_DAILY_MESSAGES) + : bumpCounter(db, `embed:${embed.bound_host}`, EMBED_DAILY_MESSAGES) +} + +/** Consume one of this host's daily recrawls. False means the cap is reached. */ +export async function consumeRecrawl(db: D1Database, boundHost: string): Promise { + return bumpCounter(db, `recrawl:${boundHost}`, EMBED_RECRAWLS_PER_DAY) } export async function touchEmbed(db: D1Database, key: string): Promise { @@ -134,3 +193,23 @@ export async function touchEmbed(db: D1Database, key: string): Promise { export function embedSnippet(key: string, origin: string): string { return `` } + +/** + * The token that proves someone holds the manage link for this embed. + * + * Derived, not stored: the same key always produces the same token, so there + * is no table, no expiry to sweep, and no second thing to keep in sync. The + * salt is a Worker secret, so a token cannot be produced from the public key + * alone. Revoking the embed is what kills the link — the row is checked first. + * + * ponytail: no login, no account. One button behind an unguessable URL is the + * whole feature. Upgrade path is the OTP flow that already exists, if the + * manage page ever does something worth stealing. + */ +export async function manageToken(key: string, salt: string): Promise { + return (await hmacHex(`manage:${key}`, salt)).slice(0, 32) +} + +export function manageUrl(key: string, token: string, origin: string): string { + return `${origin}/tools/ai-chatbot/manage?key=${key}&t=${token}` +} diff --git a/lib/tools/chatbot/schema.ts b/lib/tools/chatbot/schema.ts index b45f5fa..550c960 100644 --- a/lib/tools/chatbot/schema.ts +++ b/lib/tools/chatbot/schema.ts @@ -37,6 +37,12 @@ export const embedMessageSchema = z.object({ .default([]), }) +/** Refreshing a live embed's corpus, from the signed manage link. */ +export const recrawlSchema = z.object({ + key: z.string().max(64), + t: z.string().max(64), +}) + /** Asking for a verification code. */ export const verifySendSchema = z.object({ sessionId: z.string().uuid(), diff --git a/lib/tools/session.ts b/lib/tools/session.ts index b524584..1bcc8ee 100644 --- a/lib/tools/session.ts +++ b/lib/tools/session.ts @@ -13,13 +13,13 @@ export const REUSE_WINDOW_MS = 24 * 60 * 60 * 1000 export type SessionStatus = 'crawling' | 'ready' | 'empty' | 'failed' /** - * HMAC the visitor's IP before it touches storage. + * HMAC-SHA256 of `value` under `salt`, hex encoded. * - * A bare hash is not anonymisation — IPv4 is 2^32 addresses, which is a - * rainbow table someone can build in an afternoon. The salt is a Worker secret - * (`TOOLS_IP_SALT`), so the stored value is only linkable back by us. + * The one keyed-hash primitive for the tools. Also used for verification codes + * and embed manage tokens, so there is a single place to change if the hash + * ever moves. */ -export async function hashIp(ip: string, salt: string): Promise { +export async function hmacHex(value: string, salt: string): Promise { const enc = new TextEncoder() const key = await crypto.subtle.importKey( 'raw', @@ -28,10 +28,21 @@ export async function hashIp(ip: string, salt: string): Promise { false, ['sign'], ) - const sig = await crypto.subtle.sign('HMAC', key, enc.encode(ip)) + const sig = await crypto.subtle.sign('HMAC', key, enc.encode(value)) return [...new Uint8Array(sig)].map((b) => b.toString(16).padStart(2, '0')).join('') } +/** + * HMAC the visitor's IP before it touches storage. + * + * A bare hash is not anonymisation — IPv4 is 2^32 addresses, which is a + * rainbow table someone can build in an afternoon. The salt is a Worker secret + * (`TOOLS_IP_SALT`), so the stored value is only linkable back by us. + */ +export async function hashIp(ip: string, salt: string): Promise { + return hmacHex(ip, salt) +} + /** * Which tool a session belongs to. The table is shared across the free tools, * so this is how rows are told apart — matches the CHECK constraint in @@ -98,6 +109,17 @@ export async function savePages( ) } +/** + * Drop a session's corpus, ahead of a recrawl writing the new one. + * + * Delete then insert, not `INSERT OR REPLACE`: a page the owner has since + * removed from their site would otherwise stay in the corpus forever and the + * live bot would keep answering from it. + */ +export async function clearPages(db: D1Database, sessionId: string): Promise { + await db.prepare('DELETE FROM tool_pages WHERE session_id = ?').bind(sessionId).run() +} + /** * The most recent usable crawl of this host, if there is one. * @@ -140,6 +162,8 @@ export async function copyPages(db: D1Database, fromId: string, toId: string): P export type SessionRow = { id: string host: string + /** The address as submitted, re-validated before any recrawl reuses it. */ + target_url: string status: SessionStatus messages_used: number transcript_json: string | null @@ -149,7 +173,7 @@ export type SessionRow = { export async function getSession(db: D1Database, id: string): Promise { return db .prepare( - `SELECT id, host, status, messages_used, transcript_json, diagnosis_json + `SELECT id, host, target_url, status, messages_used, transcript_json, diagnosis_json FROM tool_sessions WHERE id = ?`, ) .bind(id) From 8991b2b32385606e5481ba7f42f596136597ad60 Mon Sep 17 00:00:00 2001 From: harshit-epyc Date: Mon, 17 Aug 2026 17:59:28 +0530 Subject: [PATCH 04/12] feat(tools): optional code reveal for environments with no email provider MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On a deployed environment the verification code only reaches a Worker log, so nobody outside the team can finish the embed claim — staging exists to be tested, and this was the one flow that could not be. `TOOLS_REVEAL_CODES=true` returns the code in the response, and the client prints it to the browser console rather than the page: it is a testing affordance, not something to show a visitor. Two conditions guard it. The flag must be set, and the send must have actually been stubbed — so configuring RESEND_API_KEY closes it off even if the flag is left behind. It cannot silently follow the code into production, where it would make email verification meaningless. Co-Authored-By: Claude Opus 5 (1M context) --- app/api/tools/chatbot/verify/route.ts | 11 ++++++++++- cloudflare-env.secrets.d.ts | 8 ++++++++ components/sections/chatbot-tool.tsx | 14 ++++++++++++-- docs/ai-chatbot-local-testing.md | 9 +++++++++ 4 files changed, 39 insertions(+), 3 deletions(-) diff --git a/app/api/tools/chatbot/verify/route.ts b/app/api/tools/chatbot/verify/route.ts index d83e170..bee1110 100644 --- a/app/api/tools/chatbot/verify/route.ts +++ b/app/api/tools/chatbot/verify/route.ts @@ -85,10 +85,19 @@ export async function POST(req: Request) { ) } + // Staging has no email provider, so the code only reaches a Worker log — + // which means nobody outside the team can finish the flow. This hands it back + // in the response instead, for a deployed environment that exists to be + // tested. Two conditions, deliberately: an explicit opt-in flag, AND the send + // having actually been stubbed. Configuring a provider closes this off even + // if someone leaves the flag set. + const reveal = stubbed && env.TOOLS_REVEAL_CODES === 'true' + return NextResponse.json({ ok: true, expiresInMinutes: CODE_TTL_MINUTES, - // Tells the UI to say where the code actually went. Never the code itself. + // Tells the UI to say where the code actually went. stubbed, + ...(reveal ? { code: issued.code } : {}), }) } diff --git a/cloudflare-env.secrets.d.ts b/cloudflare-env.secrets.d.ts index d173fd4..ac8cbc0 100644 --- a/cloudflare-env.secrets.d.ts +++ b/cloudflare-env.secrets.d.ts @@ -33,5 +33,13 @@ declare namespace Cloudflare { * Setting it also requires SPF and DKIM records on epyc.in. */ RESEND_API_KEY?: string + /** + * Returns the verification code in the API response instead of only + * logging it. **Staging only, while there is no email provider** — with it + * set, verifying an address proves nothing, so anyone reaching the page can + * claim an embed key. Only takes effect when `RESEND_API_KEY` is unset, so + * configuring a provider disables it either way. Never set in production. + */ + TOOLS_REVEAL_CODES?: string } } diff --git a/components/sections/chatbot-tool.tsx b/components/sections/chatbot-tool.tsx index 412b23a..855a606 100644 --- a/components/sections/chatbot-tool.tsx +++ b/components/sections/chatbot-tool.tsx @@ -773,13 +773,23 @@ function ClaimEmbed({ sessionId, host }: { sessionId: string; host: string }) { headers: { 'content-type': 'application/json' }, body: JSON.stringify({ action: 'send', sessionId, email: email.trim() }), }) - const body = (await res.json()) as { ok: boolean; error?: string; stubbed?: boolean } + const body = (await res.json()) as { + ok: boolean + error?: string + stubbed?: boolean + code?: string + } if (body.ok) { setStep('code') + // Only present on a build with no email provider and TOOLS_REVEAL_CODES + // set — staging, so the flow can be finished without a mailbox. Kept in + // the console rather than on the page: it is a testing affordance, not + // something to show a visitor. + if (body.code) console.info('[epyc] verification code:', body.code) setNotice( body.stubbed - ? 'Email sending is not configured yet — the code is printed in the dev server console.' + ? 'Email sending is not configured yet, so this code was written to the server log rather than sent.' : `We sent a 6-digit code to ${email.trim()}. It expires in 10 minutes.`, ) } else { diff --git a/docs/ai-chatbot-local-testing.md b/docs/ai-chatbot-local-testing.md index 49b1a6f..726c811 100644 --- a/docs/ai-chatbot-local-testing.md +++ b/docs/ai-chatbot-local-testing.md @@ -94,6 +94,15 @@ At the bottom of the report: No email is actually sent — there is no provider yet. The UI says so. + **On a deployed environment** the log is the Worker's, not your terminal: + `pnpm exec wrangler tail --env staging --search "verification code"`, or the + Logs tab in the Cloudflare dashboard. + + Easier for staging: set `TOOLS_REVEAL_CODES=true`, then open the browser + console and the code is printed there as `[epyc] verification code: 059052`. + Staging only — with it set, verifying an address proves nothing, so anyone + who reaches the page can claim an embed key. + 3. Type the 6 digits → **Verify** 4. The embed snippet appears with a copy button, and under it a **manage link**. That link is the only way to refresh the bot later — there is no button for From 562d88d474e945123551519f765e184c0cf2c579 Mon Sep 17 00:00:00 2001 From: harshit-epyc Date: Mon, 17 Aug 2026 18:27:06 +0530 Subject: [PATCH 05/12] fix(tools): fail closed without a salt, claim turns atomically, bound redirects MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three findings from review, all confirmed against the code. **The salt fallback made manage links forgeable.** Four call sites read `env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production'`. That constant is in the repo, and embed keys are public by design — they sit in the customer's page source — so anyone could compute `manageToken(key, fallback)` and recrawl a customer's site on demand. The same fallback would have made every stored `ip_hash` a plain hash of an address space small enough to enumerate, which is the opposite of what the privacy line claims. There is now one `toolsSalt()` helper returning null when unset, and all four routes return 503 rather than running on a known secret. **The per-session cap was not atomic.** `messages_used` was read at the top of the message route and incremented after the answer, so eight requests fired at once for one session all saw zero used, all called the model, and all recorded a turn. Replaced with the same conditional-update shape the daily counters already use: `reserveTurn` claims one before any model call and returns false at the cap. `releaseTurn` hands it back when the model never answered, so an outage does not cost the visitor a question. `recordTurn` becomes `saveTranscript` and no longer touches the count. **Redirects could outlast the crawl deadline.** `fetchText` re-armed the full timeout on every hop, so one page with three slow redirects could run 4 × budgetMs — up to 32s against a 20s wall clock the route promises. The budget is now an absolute deadline computed once, with each hop getting what is left. 7 tests on the reservation, including the parallel case. Note that node:sqlite executes synchronously, so that test pins the conditional-UPDATE semantics rather than reproducing true concurrency — which is the part that was wrong. Co-Authored-By: Claude Opus 5 (1M context) --- app/api/tools/chatbot/crawl/route.ts | 12 ++- app/api/tools/chatbot/embed/route.ts | 11 ++- app/api/tools/chatbot/message/route.ts | 22 +++-- app/api/tools/chatbot/recrawl/route.ts | 19 +++- app/api/tools/chatbot/verify/route.ts | 11 ++- docs/ai-chatbot-local-testing.md | 2 +- lib/crawl/fetch-pages.ts | 14 ++- lib/tools/session.test.ts | 123 +++++++++++++++++++++++++ lib/tools/session.ts | 69 ++++++++++++-- 9 files changed, 255 insertions(+), 28 deletions(-) create mode 100644 lib/tools/session.test.ts diff --git a/app/api/tools/chatbot/crawl/route.ts b/app/api/tools/chatbot/crawl/route.ts index f30414d..03a9901 100644 --- a/app/api/tools/chatbot/crawl/route.ts +++ b/app/api/tools/chatbot/crawl/route.ts @@ -15,6 +15,7 @@ import { loadPagesForScoring, savePages, saveDiagnosis, + toolsSalt, } from '@/lib/tools/session' /** @@ -45,8 +46,17 @@ export async function POST(req: Request) { const { env } = getCloudflareContext() const db = env.DB + // No salt, no crawl: `ip_hash` is the only thing standing between our storage + // and a plain record of who visited. Running on a known constant would be + // worse than not running. + const salt = toolsSalt(env) + if (!salt) { + console.error('TOOLS_IP_SALT is not set') + return NextResponse.json({ ok: false, error: 'This tool is unavailable.' }, { status: 503 }) + } + const ip = req.headers.get('cf-connecting-ip') ?? '0.0.0.0' - const ipHash = await hashIp(ip, env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production') + const ipHash = await hashIp(ip, salt) const ipKey = counterKeys.ip(ipHash) // Checked, not consumed: a crawl that fails should not cost the visitor one diff --git a/app/api/tools/chatbot/embed/route.ts b/app/api/tools/chatbot/embed/route.ts index e95496a..d361494 100644 --- a/app/api/tools/chatbot/embed/route.ts +++ b/app/api/tools/chatbot/embed/route.ts @@ -10,7 +10,7 @@ import { mintKey, } from '@/lib/tools/chatbot/embed' import { isVerified } from '@/lib/tools/chatbot/verification' -import { getSession } from '@/lib/tools/session' +import { getSession, toolsSalt } from '@/lib/tools/session' /** * Claim an embed: mint a key for a verified address and return the snippet. @@ -60,7 +60,14 @@ export async function POST(req: Request) { } const origin = new URL(req.url).origin - const salt = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + // Minting a manage link under a known salt would hand out a credential + // anyone could forge. Refuse rather than issue a worthless one. + const salt = toolsSalt(env) + if (!salt) { + console.error('TOOLS_IP_SALT is not set') + return NextResponse.json({ ok: false, error: 'This is unavailable.' }, { status: 503 }) + } // The manage link is returned with the snippet, and shown on screen rather // than emailed: there is no email provider configured yet, so a link we only diff --git a/app/api/tools/chatbot/message/route.ts b/app/api/tools/chatbot/message/route.ts index de87b59..332a1a4 100644 --- a/app/api/tools/chatbot/message/route.ts +++ b/app/api/tools/chatbot/message/route.ts @@ -10,8 +10,10 @@ import { loadPages, loadPagesForScoring, readTranscript, - recordTurn, + releaseTurn, + reserveTurn, saveDiagnosis, + saveTranscript, } from '@/lib/tools/session' /** @@ -43,17 +45,20 @@ export async function POST(req: Request) { return NextResponse.json({ ok: false, error: 'That session has expired.' }, { status: 404 }) } - // The report is the end of the conversation, not an error. - if (session.messages_used >= CAPS.messagesPerSession) { + if (session.status !== 'ready') { return NextResponse.json( - { ok: false, capped: true, error: 'You’ve used all your questions.' }, + { ok: false, error: 'There is not enough on that site to chat about.' }, { status: 409 }, ) } - if (session.status !== 'ready') { + // The report is the end of the conversation, not an error. Claimed in one + // conditional statement rather than read here and incremented after the + // answer: eight parallel requests would otherwise all see zero used, all call + // the model, and all record a turn. + if (!(await reserveTurn(db, session.id, CAPS.messagesPerSession))) { return NextResponse.json( - { ok: false, error: 'There is not enough on that site to chat about.' }, + { ok: false, capped: true, error: 'You’ve used all your questions.' }, { status: 409 }, ) } @@ -104,6 +109,9 @@ export async function POST(req: Request) { }) } catch (err) { console.error('all model tiers unavailable', err) + // The turn was claimed before the call. Nothing was answered, so hand it + // back rather than charging them a question for our outage. + await releaseTurn(db, session.id).catch(() => {}) return NextResponse.json( { ok: false, error: 'I’m having trouble reaching the model. Try again shortly.' }, { status: 503 }, @@ -142,7 +150,7 @@ export async function POST(req: Request) { { role: 'user' as const, content: parsed.data.message }, { role: 'assistant' as const, content: answer }, ] - await recordTurn(db, session.id, turns) + await saveTranscript(db, session.id, turns) } catch (err) { console.error('failed to record turn', err) } diff --git a/app/api/tools/chatbot/recrawl/route.ts b/app/api/tools/chatbot/recrawl/route.ts index 50579f1..ae5cd1d 100644 --- a/app/api/tools/chatbot/recrawl/route.ts +++ b/app/api/tools/chatbot/recrawl/route.ts @@ -11,7 +11,14 @@ import { manageToken, } from '@/lib/tools/chatbot/embed' import { scoreDeterministic } from '@/lib/tools/chatbot/diagnosis' -import { clearPages, finishSession, getSession, savePages, saveDiagnosis } from '@/lib/tools/session' +import { + clearPages, + finishSession, + getSession, + savePages, + saveDiagnosis, + toolsSalt, +} from '@/lib/tools/session' /** * Re-read a live embed's site and replace its corpus. @@ -37,7 +44,15 @@ export async function POST(req: Request) { const { env } = getCloudflareContext() const db = env.DB - const salt = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + // The manage token is the only credential on this route, and the embed key it + // is derived from is public. A known fallback salt would make it forgeable by + // anyone who can read the customer's page source. + const salt = toolsSalt(env) + if (!salt) { + console.error('TOOLS_IP_SALT is not set') + return NextResponse.json({ ok: false, error: 'This is unavailable.' }, { status: 503 }) + } const embed = await findEmbedByKey(db, parsed.data.key) if (!embed || embed.status !== 'active') { diff --git a/app/api/tools/chatbot/verify/route.ts b/app/api/tools/chatbot/verify/route.ts index bee1110..c1cee63 100644 --- a/app/api/tools/chatbot/verify/route.ts +++ b/app/api/tools/chatbot/verify/route.ts @@ -7,7 +7,7 @@ import { checkCode, issueCode, } from '@/lib/tools/chatbot/verification' -import { getSession } from '@/lib/tools/session' +import { getSession, toolsSalt } from '@/lib/tools/session' /** * Send a verification code, and check one. @@ -21,7 +21,14 @@ export async function POST(req: Request) { const { env } = getCloudflareContext() const db = env.DB - const pepper = env.TOOLS_IP_SALT ?? 'dev-salt-not-for-production' + + // Codes are stored as an HMAC under this. A published fallback would mean + // stored hashes are reversible by brute force over a million six-digit codes. + const pepper = toolsSalt(env) + if (!pepper) { + console.error('TOOLS_IP_SALT is not set') + return NextResponse.json({ ok: false, error: 'This is unavailable.' }, { status: 503 }) + } if (json?.action === 'check') { const parsed = verifyCheckSchema.safeParse(json) diff --git a/docs/ai-chatbot-local-testing.md b/docs/ai-chatbot-local-testing.md index 726c811..1a1ef52 100644 --- a/docs/ai-chatbot-local-testing.md +++ b/docs/ai-chatbot-local-testing.md @@ -18,7 +18,7 @@ TOOLS_SESSIONS_PER_IP=100 | Key | Why | |---|---| | `OPENROUTER_API_KEY` | The only one you cannot invent. Get it from openrouter.ai | -| `TOOLS_IP_SALT` | Hashes visitor IPs, verification codes, and embed manage links. Any string locally | +| `TOOLS_IP_SALT` | **Required.** Hashes visitor IPs, verification codes, and embed manage links. Any string locally, but without it every tool route returns 503 — there is deliberately no fallback, since a known salt would make manage links forgeable and stored IP hashes reversible | | `TOOLS_SESSIONS_PER_IP` | Local only. Without it you get 3 crawls a day, because localhost has no per-visitor IP and everything shares one counter | **`TOOLS_SESSIONS_PER_IP` must stay unset in staging and production.** It diff --git a/lib/crawl/fetch-pages.ts b/lib/crawl/fetch-pages.ts index 49d682f..6f9e343 100644 --- a/lib/crawl/fetch-pages.ts +++ b/lib/crawl/fetch-pages.ts @@ -222,15 +222,23 @@ async function fetchText( ): Promise { if (budgetMs <= 0) return null + // The budget covers the whole chain, not each hop. Re-arming the full timeout + // per redirect let one page take (maxRedirects + 1) × budgetMs — up to 32s + // against a 20s crawl deadline, which is the wall-clock cap the route + // promises. The caller still decides the size of the budget; a sitemap is + // worth waiting longer for than a page. + const deadline = Date.now() + budgetMs + let current = url for (let hop = 0; hop <= LIMITS.maxRedirects; hop++) { const checked = validateUrl(current) if (!checked.ok) return null - // The caller decides the budget — a sitemap is worth waiting longer for - // than a page. Clamping here would silently override that. + const remaining = deadline - Date.now() + if (remaining <= 0) return null + const controller = new AbortController() - const timer = setTimeout(() => controller.abort(), budgetMs) + const timer = setTimeout(() => controller.abort(), remaining) try { const res = await doFetch(checked.url, { diff --git a/lib/tools/session.test.ts b/lib/tools/session.test.ts new file mode 100644 index 0000000..4139b67 --- /dev/null +++ b/lib/tools/session.test.ts @@ -0,0 +1,123 @@ +import { describe, expect, it } from 'vitest' +import { DatabaseSync } from 'node:sqlite' +import { releaseTurn, reserveTurn } from './session' + +/** + * The per-session message cap is the only thing bounding how many model calls + * one visitor can make, and it used to be a read here, an increment later — so + * eight requests fired at once all saw zero used. What is at risk is the + * semantics of one conditional UPDATE, which is engine behaviour, so this runs + * the real SQL against node:sqlite exactly as counters.test.ts does. + */ + +function fakeD1() { + const db = new DatabaseSync(':memory:') + db.exec(` + CREATE TABLE tool_sessions ( + id TEXT PRIMARY KEY, + messages_used INTEGER NOT NULL DEFAULT 0, + transcript_json TEXT + ); + INSERT INTO tool_sessions (id, messages_used) VALUES ('s1', 0), ('s2', 0); + `) + + const handle = { + prepare(sql: string) { + const stmt = db.prepare(sql) + return { + bind(...values: unknown[]) { + return { + async run() { + const r = stmt.run(...(values as never[])) + return { meta: { changes: Number(r.changes) } } + }, + } + }, + } + }, + } as unknown as D1Database + + const used = (id: string) => + (db.prepare('SELECT messages_used FROM tool_sessions WHERE id = ?').get(id) as { + messages_used: number + }).messages_used + + return { handle, used } +} + +const LIMIT = 8 + +describe('reserveTurn', () => { + it('allows exactly the limit, then refuses and stays refused', async () => { + const { handle, used } = fakeD1() + + for (let i = 0; i < LIMIT; i++) { + expect(await reserveTurn(handle, 's1', LIMIT)).toBe(true) + } + + expect(used('s1')).toBe(LIMIT) + expect(await reserveTurn(handle, 's1', LIMIT)).toBe(false) + expect(await reserveTurn(handle, 's1', LIMIT)).toBe(false) + + // A refused claim must not have incremented anything on its way out. + expect(used('s1')).toBe(LIMIT) + }) + + it('holds when claims arrive at once rather than in sequence', async () => { + const { handle, used } = fakeD1() + + // The failure this replaced: every one of these read messages_used = 0 + // before any of them wrote, so all twenty were allowed. + const claims = await Promise.all( + Array.from({ length: 20 }, () => reserveTurn(handle, 's1', LIMIT)), + ) + + expect(claims.filter(Boolean)).toHaveLength(LIMIT) + expect(used('s1')).toBe(LIMIT) + }) + + it('counts each session separately', async () => { + const { handle, used } = fakeD1() + + for (let i = 0; i < LIMIT; i++) await reserveTurn(handle, 's1', LIMIT) + + expect(await reserveTurn(handle, 's2', LIMIT)).toBe(true) + expect(used('s2')).toBe(1) + }) + + it('refuses an unknown session rather than creating one', async () => { + const { handle } = fakeD1() + expect(await reserveTurn(handle, 'nope', LIMIT)).toBe(false) + }) +}) + +describe('releaseTurn', () => { + it('hands back a claim when the model never answered', async () => { + const { handle, used } = fakeD1() + + await reserveTurn(handle, 's1', LIMIT) + await reserveTurn(handle, 's1', LIMIT) + await releaseTurn(handle, 's1') + + expect(used('s1')).toBe(1) + }) + + it('reopens the session when a release follows the last claim', async () => { + const { handle } = fakeD1() + + for (let i = 0; i < LIMIT; i++) await reserveTurn(handle, 's1', LIMIT) + expect(await reserveTurn(handle, 's1', LIMIT)).toBe(false) + + await releaseTurn(handle, 's1') + expect(await reserveTurn(handle, 's1', LIMIT)).toBe(true) + }) + + it('never drops below zero', async () => { + const { handle, used } = fakeD1() + + await releaseTurn(handle, 's1') + await releaseTurn(handle, 's1') + + expect(used('s1')).toBe(0) + }) +}) diff --git a/lib/tools/session.ts b/lib/tools/session.ts index 1bcc8ee..c8f4823 100644 --- a/lib/tools/session.ts +++ b/lib/tools/session.ts @@ -12,6 +12,22 @@ export const REUSE_WINDOW_MS = 24 * 60 * 60 * 1000 export type SessionStatus = 'crawling' | 'ready' | 'empty' | 'failed' +/** + * The HMAC salt, or null when it is not configured. + * + * There was a `?? 'dev-salt-not-for-production'` fallback at all four call + * sites. That is a published constant, and embed keys are public by design, so + * anyone could compute `manageToken(key, fallback)` and recrawl a customer's + * site — and every stored `ip_hash` would be a plain hash of an address space + * small enough to enumerate. A missing salt now fails closed instead: the + * caller returns 503 rather than quietly running on a known secret. + * + * Local development sets it in `.dev.vars` — see docs/ai-chatbot-local-testing.md. + */ +export function toolsSalt(env: { TOOLS_IP_SALT?: string }): string | null { + return env.TOOLS_IP_SALT || null +} + /** * HMAC-SHA256 of `value` under `salt`, hex encoded. * @@ -192,23 +208,56 @@ export function readTranscript(row: SessionRow): Turn[] { } /** - * Record one exchange and consume one of the session's messages. + * Claim one of the session's messages, before any model call is made. + * + * Returns false when the cap is already reached. This has to be the same + * conditional-update shape as the daily counters: reading `messages_used` and + * incrementing it later is not atomic, so N requests fired at once for one + * session all read the same value, all call the model, and all record a turn — + * eight becomes however many the client sends in parallel. + * + * Reserving up front means a turn can be consumed by a request that then fails. + * `releaseTurn` hands it back for the one case we can detect. + */ +export async function reserveTurn( + db: D1Database, + id: string, + limit: number, +): Promise { + const res = await db + .prepare( + `UPDATE tool_sessions SET messages_used = messages_used + 1 + WHERE id = ? AND messages_used < ?`, + ) + .bind(id, limit) + .run() + + return (res.meta.changes ?? 0) > 0 +} + +/** Give back a reserved turn when the model never answered. */ +export async function releaseTurn(db: D1Database, id: string): Promise { + await db + .prepare('UPDATE tool_sessions SET messages_used = messages_used - 1 WHERE id = ? AND messages_used > 0') + .bind(id) + .run() +} + +/** + * Store the conversation so far. * - * Single statement so the count cannot drift from the transcript: if the write - * fails, neither happened. Storing the transcript tells us which questions - * visitors actually ask, which is the feedback loop for refining the ten. + * The count is no longer incremented here — `reserveTurn` owns it, because the + * cap has to be enforced before the model call rather than after it. Storing + * the transcript tells us which questions visitors actually ask, which is the + * feedback loop for refining the ten. */ -export async function recordTurn( +export async function saveTranscript( db: D1Database, id: string, transcript: Turn[], ): Promise { await db - .prepare( - `UPDATE tool_sessions - SET transcript_json = ?, messages_used = messages_used + 1 - WHERE id = ?`, - ) + .prepare('UPDATE tool_sessions SET transcript_json = ? WHERE id = ?') .bind(JSON.stringify(transcript), id) .run() } From c7f2a85e981fd063eda0e9422028a7a3ce628b29 Mon Sep 17 00:00:00 2001 From: Keshav Sharma Date: Sun, 23 Aug 2026 16:18:27 +0530 Subject: [PATCH 06/12] feat(cms): provider abstraction, Payload provider, and a parity checker Routes, metadata, sitemap and the markdown builders now read through a CMS-neutral provider interface instead of importing Strapi shapes directly, with StrapiProvider and PayloadProvider behind it. Provider selection stays an environment setting so the rollback is a variable, not a revert. Draft reads require CMS_MODE=draft and DEPLOYMENT_ROLE=preview together, and anything else falls back to published, so a single mistyped variable cannot expose drafts on the public domain. A preview deployment additionally serves noindex, no-store, and an empty sitemap. /api/cms/revalidate accepts signed webhooks from Payload, mapping only known collections onto application-owned paths and purging the provider's fetch tags alongside them. Draft saves are ignored on production deployments. scripts/cms-parity.ts compares what the website would render from each provider, on normalised output rather than raw API responses, so a reported difference is one a visitor could see. Differences the migration intends are declared and reported separately from defects: media URLs moving from Strapi derivatives at bare paths to originals at absolute URLs (Cloudflare resizing verified live on both forms), folded duplicate type spellings, and the corrected e-commerce slug. Workflow and wrangler config carry the new variables per environment. Staging becomes the preview deployment; both environments keep CMS_PROVIDER=strapi until cutover. Co-Authored-By: Claude Opus 5 (1M context) --- .env.example | 24 +- .github/workflows/deploy-production.yml | 17 +- .github/workflows/deploy-staging.yml | 17 +- .gitignore | 3 + app/(my-app)/blog/[slug]/page.tsx | 36 +- app/(my-app)/blog/page.tsx | 12 +- app/(my-app)/gallery/[slug]/page.tsx | 32 +- app/(my-app)/gallery/page.tsx | 10 +- app/(my-app)/layout.tsx | 5 +- app/(my-app)/projects/page.tsx | 13 +- app/(my-app)/sitemap.ts | 16 +- app/api/cms/revalidate/route.ts | 35 ++ app/robots.ts | 4 + docs/payload-cms-migration-plan.md | 744 ++++++++++++++++++++++++ lib/blogs/normalise.ts | 6 +- lib/cms/config.test.ts | 21 + lib/cms/config.ts | 18 + lib/cms/index.ts | 16 + lib/cms/payload-provider.test.ts | 127 ++++ lib/cms/payload-provider.ts | 140 +++++ lib/cms/revalidation.test.ts | 40 ++ lib/cms/revalidation.ts | 61 ++ lib/cms/strapi-provider.ts | 96 +++ lib/cms/types.ts | 74 +++ lib/gallery/normalise.ts | 16 +- lib/markdown/sources.ts | 51 +- lib/projects/normalise.ts | 10 +- lib/strapi/client.ts | 4 +- middleware.ts | 15 +- package.json | 3 +- scripts/cms-parity.ts | 205 +++++++ wrangler.jsonc | 27 + 32 files changed, 1734 insertions(+), 164 deletions(-) create mode 100644 app/api/cms/revalidate/route.ts create mode 100644 docs/payload-cms-migration-plan.md create mode 100644 lib/cms/config.test.ts create mode 100644 lib/cms/config.ts create mode 100644 lib/cms/index.ts create mode 100644 lib/cms/payload-provider.test.ts create mode 100644 lib/cms/payload-provider.ts create mode 100644 lib/cms/revalidation.test.ts create mode 100644 lib/cms/revalidation.ts create mode 100644 lib/cms/strapi-provider.ts create mode 100644 lib/cms/types.ts create mode 100644 scripts/cms-parity.ts diff --git a/.env.example b/.env.example index 832b942..bfd8252 100644 --- a/.env.example +++ b/.env.example @@ -1,16 +1,30 @@ -# Strapi v5 server base URL (no trailing slash and no /admin path) +# Provider switch retained during the rollback window. Unknown values use Strapi. +CMS_PROVIDER=strapi + +# Only this exact pair permits draft reads. All other values are published-only. +CMS_MODE=published +DEPLOYMENT_ROLE=production + +# Payload API base URL and server-only read credentials. +PAYLOAD_URL=https://cms.epyc.in +PAYLOAD_READ_TOKEN= +PAYLOAD_PREVIEW_TOKEN= + +# HMAC secret shared with Payload's revalidation webhook. The webhook sends a +# hex SHA-256 signature in the x-epyc-signature header. +CMS_REVALIDATION_SECRET= + +# Strapi v5 server base URL retained for migration and rollback. STRAPI_URL=https://your-strapi-server.com # Read-only API token from Strapi Admin → Settings → API Tokens # Leave blank if the collections are publicly readable STRAPI_API_TOKEN= -# Set to "true" to fetch unpublished drafts (?status=draft) with caching off. -# Use only on preview/staging — the token above must have draft read access. +# Legacy fallback. Drafts still require DEPLOYMENT_ROLE=preview. STRAPI_PREVIEW= -# Public base URL of the R2 bucket (custom domain) that serves Strapi media. -# The image loader prepends this to Strapi's bare /uploads/* paths. +# Public base URL of the R2 media bucket. NEXT_PUBLIC_MEDIA_BASE_URL=https://media.epyc.in # Contact form submissions are stored in Cloudflare D1, reached through the `DB` diff --git a/.github/workflows/deploy-production.yml b/.github/workflows/deploy-production.yml index 014f263..4986d7b 100644 --- a/.github/workflows/deploy-production.yml +++ b/.github/workflows/deploy-production.yml @@ -45,6 +45,15 @@ jobs: STRAPI_URL: ${{ secrets.PRODUCTION_STRAPI_URL }} STRAPI_API_TOKEN: ${{ secrets.PRODUCTION_STRAPI_API_TOKEN }} STRAPI_PREVIEW: ${{ secrets.PRODUCTION_STRAPI_PREVIEW }} + # Published-only reads. robots.ts and the root layout's metadata are + # evaluated at build time, so these must be present here as well as in + # the Worker's vars. + CMS_PROVIDER: strapi + CMS_MODE: published + DEPLOYMENT_ROLE: production + PAYLOAD_URL: https://cms.epyc.in + PAYLOAD_READ_TOKEN: ${{ secrets.PRODUCTION_PAYLOAD_READ_TOKEN }} + CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} - name: Upload secrets to Cloudflare Workers (production) env: @@ -53,15 +62,21 @@ jobs: STRAPI_URL: ${{ secrets.PRODUCTION_STRAPI_URL }} STRAPI_API_TOKEN: ${{ secrets.PRODUCTION_STRAPI_API_TOKEN }} STRAPI_PREVIEW: ${{ secrets.PRODUCTION_STRAPI_PREVIEW }} + PAYLOAD_READ_TOKEN: ${{ secrets.PRODUCTION_PAYLOAD_READ_TOKEN }} + CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} run: | jq -n \ --arg STRAPI_URL "$STRAPI_URL" \ --arg STRAPI_API_TOKEN "$STRAPI_API_TOKEN" \ --arg STRAPI_PREVIEW "$STRAPI_PREVIEW" \ + --arg PAYLOAD_READ_TOKEN "$PAYLOAD_READ_TOKEN" \ + --arg CMS_REVALIDATION_SECRET "$CMS_REVALIDATION_SECRET" \ '{ STRAPI_URL: $STRAPI_URL, STRAPI_API_TOKEN: $STRAPI_API_TOKEN, - STRAPI_PREVIEW: $STRAPI_PREVIEW + STRAPI_PREVIEW: $STRAPI_PREVIEW, + PAYLOAD_READ_TOKEN: $PAYLOAD_READ_TOKEN, + CMS_REVALIDATION_SECRET: $CMS_REVALIDATION_SECRET }' \ | pnpm exec wrangler secret bulk --env production diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 46ab5a6..75a0d88 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -45,6 +45,15 @@ jobs: STRAPI_URL: ${{ secrets.STAGING_STRAPI_URL }} STRAPI_API_TOKEN: ${{ secrets.STAGING_STRAPI_API_TOKEN }} STRAPI_PREVIEW: ${{ secrets.STAGING_STRAPI_PREVIEW }} + # Staging is the content-preview site: draft reads, noindex, no sitemap. + # robots.ts and the root layout's metadata are evaluated at build time, + # so these must be present here as well as in the Worker's vars. + CMS_PROVIDER: strapi + CMS_MODE: draft + DEPLOYMENT_ROLE: preview + PAYLOAD_URL: https://cms.epyc.in + PAYLOAD_PREVIEW_TOKEN: ${{ secrets.STAGING_PAYLOAD_PREVIEW_TOKEN }} + CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} - name: Upload secrets to Cloudflare Workers (staging) env: @@ -53,15 +62,21 @@ jobs: STRAPI_URL: ${{ secrets.STAGING_STRAPI_URL }} STRAPI_API_TOKEN: ${{ secrets.STAGING_STRAPI_API_TOKEN }} STRAPI_PREVIEW: ${{ secrets.STAGING_STRAPI_PREVIEW }} + PAYLOAD_PREVIEW_TOKEN: ${{ secrets.STAGING_PAYLOAD_PREVIEW_TOKEN }} + CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} run: | jq -n \ --arg STRAPI_URL "$STRAPI_URL" \ --arg STRAPI_API_TOKEN "$STRAPI_API_TOKEN" \ --arg STRAPI_PREVIEW "$STRAPI_PREVIEW" \ + --arg PAYLOAD_PREVIEW_TOKEN "$PAYLOAD_PREVIEW_TOKEN" \ + --arg CMS_REVALIDATION_SECRET "$CMS_REVALIDATION_SECRET" \ '{ STRAPI_URL: $STRAPI_URL, STRAPI_API_TOKEN: $STRAPI_API_TOKEN, - STRAPI_PREVIEW: $STRAPI_PREVIEW + STRAPI_PREVIEW: $STRAPI_PREVIEW, + PAYLOAD_PREVIEW_TOKEN: $PAYLOAD_PREVIEW_TOKEN, + CMS_REVALIDATION_SECRET: $CMS_REVALIDATION_SECRET }' \ | pnpm exec wrangler secret bulk --env staging diff --git a/.gitignore b/.gitignore index 7e5dc36..fd0c7a4 100644 --- a/.gitignore +++ b/.gitignore @@ -64,3 +64,6 @@ graphify-out/ # Local-only embed widget test page — never ship this public/embed-test.html + +# Parity and migration reports — generated, may contain CMS content +artifacts/ diff --git a/app/(my-app)/blog/[slug]/page.tsx b/app/(my-app)/blog/[slug]/page.tsx index ca8b0c5..7acb2a8 100644 --- a/app/(my-app)/blog/[slug]/page.tsx +++ b/app/(my-app)/blog/[slug]/page.tsx @@ -1,7 +1,6 @@ import type { Metadata } from 'next' import { notFound } from 'next/navigation' -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiList, StrapiBlog } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { BlogPost } from '@/components/sections/blog-post' import { CTAFooter } from '@/components/sections/cta-footer' import { normalise } from '@/lib/blogs/normalise' @@ -40,17 +39,7 @@ function rewriteMediaUrls(html: string): string { export async function generateMetadata({ params }: { params: Params }): Promise { const { slug } = await params - const { data } = await fetchStrapi>('/blogs', { - 'filters[slug][$eq]': slug, - 'fields[0]': 'title', - 'fields[1]': 'metaTitle', - 'fields[2]': 'metaDescription', - 'fields[3]': 'content', - 'fields[4]': 'coverImageAlt', - 'populate[coverImage][fields]': 'url,width,height,alternativeText', - 'pagination[limit]': '1', - }) - const blog = data[0] + const blog = await getCMS().getBlogBySlug(slug) if (!blog) return {} const ogImage = blog.coverImage @@ -73,26 +62,13 @@ export async function generateMetadata({ params }: { params: Params }): Promise< export default async function BlogDetailPage({ params }: { params: Params }) { const { slug } = await params - const [{ data }, { data: relatedData }] = await Promise.all([ - fetchStrapi>('/blogs', { - 'filters[slug][$eq]': slug, - 'populate[coverImage][fields]': 'url,width,height,alternativeText,formats', - 'populate[author][fields]': 'name,slug', - 'pagination[limit]': '1', - }), - fetchStrapi>('/blogs', { - 'filters[slug][$ne]': slug, - 'populate[coverImage][fields]': 'url,width,height,alternativeText,formats', - 'populate[author][fields]': 'name,slug', - 'sort': 'publishedDate:desc', - 'pagination[limit]': '3', - }), + const [blog, relatedData] = await Promise.all([ + getCMS().getBlogBySlug(slug), + getCMS().listBlogs({ excludeSlug: slug, limit: 3 }), ]) - - const blog = data[0] if (!blog) notFound() - const relatedBlogs = relatedData.filter((b) => b.slug).map((b) => normalise(b)) + const relatedBlogs = relatedData.map((b) => normalise(b)) const jsonLd = { '@context': 'https://schema.org', diff --git a/app/(my-app)/blog/page.tsx b/app/(my-app)/blog/page.tsx index 8fbb771..58e1d27 100644 --- a/app/(my-app)/blog/page.tsx +++ b/app/(my-app)/blog/page.tsx @@ -1,6 +1,5 @@ import type { Metadata } from 'next' -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiList, StrapiBlog } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { BlogIndex } from '@/components/sections/blog-index' import { CTAFooter } from '@/components/sections/cta-footer' import { normalise } from '@/lib/blogs/normalise' @@ -19,13 +18,8 @@ export const metadata: Metadata = { export const revalidate = 60 export default async function BlogsPage() { - const { data } = await fetchStrapi>('/blogs', { - 'populate[coverImage][fields]': 'url,width,height,alternativeText,formats', - 'populate[author][fields]': 'name,slug', - 'sort': 'publishedDate:desc', - 'pagination[limit]': '100', - }) - const blogs = data.filter((b) => b.slug).map((b) => normalise(b)) + const data = await getCMS().listBlogs({ limit: 100 }) + const blogs = data.map((b) => normalise(b)) const itemListJsonLd = { '@context': 'https://schema.org', diff --git a/app/(my-app)/gallery/[slug]/page.tsx b/app/(my-app)/gallery/[slug]/page.tsx index d2d6ba5..41673d9 100644 --- a/app/(my-app)/gallery/[slug]/page.tsx +++ b/app/(my-app)/gallery/[slug]/page.tsx @@ -1,7 +1,6 @@ import type { Metadata } from 'next' import { notFound } from 'next/navigation' -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiList, StrapiGalleryItem } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { normaliseGallery } from '@/lib/gallery/normalise' import { GalleryDetail } from '@/components/sections/gallery-detail' import { CTAFooter } from '@/components/sections/cta-footer' @@ -11,22 +10,13 @@ export const revalidate = 60 const GALLERY_TITLE = 'Gallery' const GALLERY_DESCRIPTION = 'Stills, motion clips, and prototypes from the EPYC studio.' -const POPULATE_PARAMS = { - 'populate[image][fields]': 'url,width,height,alternativeText', -} - export async function generateMetadata({ params, }: { params: Promise<{ slug: string }> }): Promise { const { slug } = await params - const { data } = await fetchStrapi>('/gallery-items', { - 'filters[slug][$eq]': slug, - ...POPULATE_PARAMS, - 'pagination[limit]': '1', - }) - const raw = data[0] + const raw = await getCMS().getGalleryItemBySlug(slug) if (!raw) return { title: GALLERY_TITLE } const item = normaliseGallery(raw) @@ -60,24 +50,14 @@ export default async function GalleryItemPage({ }) { const { slug } = await params - const [{ data }, { data: allData }] = await Promise.all([ - fetchStrapi>('/gallery-items', { - 'filters[slug][$eq]': slug, - ...POPULATE_PARAMS, - 'pagination[limit]': '1', - }), - fetchStrapi>('/gallery-items', { - 'filters[slug][$ne]': slug, - ...POPULATE_PARAMS, - 'pagination[limit]': '3', - }), + const [raw, allData] = await Promise.all([ + getCMS().getGalleryItemBySlug(slug), + getCMS().listGalleryItems({ excludeSlug: slug, limit: 3 }), ]) - - const raw = data[0] if (!raw) notFound() const item = normaliseGallery(raw) - const related = allData.filter((r) => r.slug).map((r) => normaliseGallery(r)) + const related = allData.map((r) => normaliseGallery(r)) return ( <> diff --git a/app/(my-app)/gallery/page.tsx b/app/(my-app)/gallery/page.tsx index 05ba4ba..055fa7a 100644 --- a/app/(my-app)/gallery/page.tsx +++ b/app/(my-app)/gallery/page.tsx @@ -1,6 +1,5 @@ import type { Metadata } from 'next' -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiList, StrapiGalleryItem } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { normaliseGallery } from '@/lib/gallery/normalise' import { GalleryIndex } from '@/components/sections/gallery-index' import { FAQs } from '@/components/sections/faqs' @@ -19,11 +18,8 @@ export const metadata: Metadata = { export const revalidate = 60 export default async function GalleryPage() { - const { data } = await fetchStrapi>('/gallery-items', { - 'populate[image][fields]': 'url,width,height,alternativeText', - 'pagination[limit]': '500', - }) - const items = data.filter((item) => item.slug).map((item) => normaliseGallery(item)) + const data = await getCMS().listGalleryItems({ limit: 500 }) + const items = data.map((item) => normaliseGallery(item)) return ( <> diff --git a/app/(my-app)/layout.tsx b/app/(my-app)/layout.tsx index 58247c6..c0ebe76 100644 --- a/app/(my-app)/layout.tsx +++ b/app/(my-app)/layout.tsx @@ -5,6 +5,7 @@ import Script from 'next/script' import './globals.css' import { FloatingMenuButton } from '@/components/ui/floating-menu' import { site } from '@/data/site' +import { isPreviewDeployment } from '@/lib/cms/config' const inter = Inter({ subsets: ['latin'], @@ -89,7 +90,9 @@ export const metadata: Metadata = { twitter: { card: 'summary_large_image', }, - robots: { index: true, follow: true, 'max-image-preview': 'large' }, + robots: isPreviewDeployment() + ? { index: false, follow: false } + : { index: true, follow: true, 'max-image-preview': 'large' }, } export default function RootLayout({ diff --git a/app/(my-app)/projects/page.tsx b/app/(my-app)/projects/page.tsx index 0b05933..679e13f 100644 --- a/app/(my-app)/projects/page.tsx +++ b/app/(my-app)/projects/page.tsx @@ -1,6 +1,5 @@ import type { Metadata } from 'next' -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiList, StrapiProject } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { ProjectsIndex } from '@/components/sections/projects-index' import { CTAFooter } from '@/components/sections/cta-footer' import { normaliseProject } from '@/lib/projects/normalise' @@ -18,14 +17,8 @@ export const metadata: Metadata = { export const revalidate = 60 export default async function ProjectsPage() { - const { data } = await fetchStrapi>('/projects', { - 'populate[thumbnail][fields]': 'url,width,height,alternativeText,formats', - 'populate[industry][fields]': 'title,slug', - 'populate[platform][fields]': 'title,slug', - 'sort': 'featured:desc,publishedAt:desc', - 'pagination[limit]': '200', - }) - const projects = data.filter((p) => p.slug).map((p) => normaliseProject(p)) + const data = await getCMS().listProjects({ limit: 200 }) + const projects = data.map((p) => normaliseProject(p)) return ( <> diff --git a/app/(my-app)/sitemap.ts b/app/(my-app)/sitemap.ts index 06574d2..e807a42 100644 --- a/app/(my-app)/sitemap.ts +++ b/app/(my-app)/sitemap.ts @@ -1,23 +1,17 @@ import type { MetadataRoute } from "next"; import { site } from "@/data/site"; -import { fetchStrapi } from "@/lib/strapi/client"; - -type SlugEntry = { slug: string; publishedAt: string }; -type StrapiSlugList = { data: SlugEntry[]; meta: unknown }; +import { getCMS } from "@/lib/cms"; +import { isPreviewDeployment } from "@/lib/cms/config"; export const revalidate = 60; export default async function sitemap(): Promise { + if (isPreviewDeployment()) return []; const url = (path: string) => ({ url: `${site.url}${path}` }); - const blogs = await fetchStrapi("/blogs", { - "fields[0]": "slug", - "fields[1]": "publishedAt", - "pagination[limit]": "1000", - "sort": "publishedDate:desc", - }); + const blogs = await getCMS().listBlogSlugsForSitemap(); - const blogEntries = blogs.data.map(({ slug, publishedAt }) => ({ + const blogEntries = blogs.map(({ slug, publishedAt }) => ({ url: `${site.url}/blog/${slug}`, lastModified: new Date(publishedAt), })); diff --git a/app/api/cms/revalidate/route.ts b/app/api/cms/revalidate/route.ts new file mode 100644 index 0000000..d4f4a86 --- /dev/null +++ b/app/api/cms/revalidate/route.ts @@ -0,0 +1,35 @@ +import { revalidatePath, revalidateTag } from 'next/cache' +import { isPreviewDeployment } from '@/lib/cms/config' +import { parseRevalidationEvent, pathsForEvent, tagsForEvent, verifyWebhookSignature } from '@/lib/cms/revalidation' + +export async function POST(request: Request) { + const startedAt = Date.now() + const body = await request.text() + const valid = await verifyWebhookSignature(body, request.headers.get('x-epyc-signature'), process.env.CMS_REVALIDATION_SECRET ?? '') + if (!valid) return Response.json({ error: 'Invalid signature' }, { status: 401 }) + + let parsed: unknown + try { + parsed = JSON.parse(body) + } catch { + return Response.json({ error: 'Invalid JSON' }, { status: 400 }) + } + const event = parseRevalidationEvent(parsed) + if (!event) return Response.json({ error: 'Invalid event' }, { status: 400 }) + + // Draft saves affect preview only. Publishing, unpublishing and deletion are + // sent to both frontends by Payload and accepted by either deployment. + if (event.action === 'draft' && !isPreviewDeployment()) { + return Response.json({ eventId: event.eventId, revalidated: [], ignored: 'draft-on-production' }) + } + + const paths = pathsForEvent(event) + const tags = tagsForEvent(event) + for (const path of paths) revalidatePath(path) + // `expire: 0` drops the entry now, so the first request after a publish gets + // the new content. The recommended `'max'` profile is stale-while-revalidate, + // which would serve the pre-publish version to that first visitor. + for (const tag of tags) revalidateTag(tag, { expire: 0 }) + console.info('CMS revalidation', { eventId: event.eventId, collection: event.collection, action: event.action, paths, tags, durationMs: Date.now() - startedAt }) + return Response.json({ eventId: event.eventId, revalidated: paths, tags }) +} diff --git a/app/robots.ts b/app/robots.ts index da133dc..883f89c 100644 --- a/app/robots.ts +++ b/app/robots.ts @@ -1,7 +1,11 @@ import type { MetadataRoute } from "next"; import { site } from "@/data/site"; +import { isPreviewDeployment } from "@/lib/cms/config"; export default function robots(): MetadataRoute.Robots { + if (isPreviewDeployment()) { + return { rules: { userAgent: "*", disallow: "/" } }; + } return { rules: [ { diff --git a/docs/payload-cms-migration-plan.md b/docs/payload-cms-migration-plan.md new file mode 100644 index 0000000..8f6ff06 --- /dev/null +++ b/docs/payload-cms-migration-plan.md @@ -0,0 +1,744 @@ +# Strapi to Payload CMS Migration Plan + +## 1. Objective + +Replace Strapi with Payload CMS without changing public URLs, rendered content, +media availability, SEO output, editorial workflow, or the reliability of the +public website. + +The selected architecture uses: + +- one Payload CMS deployment; +- one production Payload D1 database; +- one production R2 media store; +- one protected preview website that reads the latest drafts; and +- one public website that reads published versions only. + +Editors write each document once. Saving a draft updates the preview website. +Publishing the same document updates the public website. + +This plan covers the website repository and the current Strapi repository at: + +```text +/Users/keshavsharma/Documents/Cloned Repos/epyc-website-nextjs +/Users/keshavsharma/Documents/Cloned Repos/epyc-strapi-cms +``` + +## 2. Target Architecture + +```text + Editors + | + v + +---------------------+ + | Payload Admin / API | + | cms.epyc.in | + +----------+----------+ + | + +-----------+-----------+ + | | + v v + +----------------------+ +----------------------+ + | PAYLOAD_DB | | R2 media | + | dedicated Cloudflare | | images and files | + | D1 database | | | + +----------------------+ +----------------------+ + | + +----------------+----------------+ + | | + v v + +-------------------------+ +-------------------------+ + | Protected preview site | | Public production site | + | latest draft versions | | published versions only | + | noindex, no shared cache| | ISR + webhook refresh | + +-------------------------+ +-------------------------+ +``` + +Payload should run as its own OpenNext/Cloudflare Worker, not inside a +Cloudflare Container and not inside the public website Worker. The official +Payload D1 adapter consumes a native Worker D1 binding. Keeping the CMS and +website deployments separate also prevents a CMS deployment failure from +taking down the public site. + +The preview website may reuse the existing staging website Worker and domain, +but its role becomes "content preview," not a separate CMS environment. Both +websites read from the same Payload database. + +## 3. Environment and Resource Layout + +### Payload service + +Suggested resources: + +```text +Worker: epyc-payload-cms +Domain: cms.epyc.in +D1 database: epyc-payload-production +D1 binding: PAYLOAD_DB +R2 bucket: existing production media bucket, or a dedicated Payload bucket +``` + +Required secrets and variables include: + +```text +PAYLOAD_SECRET +PAYLOAD_PUBLIC_SERVER_URL=https://cms.epyc.in +PREVIEW_SITE_URL=https:// +PREVIEW_SECRET +R2_BUCKET +R2_ENDPOINT +R2_ACCESS_KEY_ID +R2_SECRET_ACCESS_KEY +R2_PUBLIC_URL +``` + +Exact names can follow the selected Payload storage adapter and deployment +template, but secrets must never use a `NEXT_PUBLIC_` prefix. + +### Public website + +```text +CMS_PROVIDER=payload +CMS_MODE=published +DEPLOYMENT_ROLE=production +PAYLOAD_URL=https://cms.epyc.in +PAYLOAD_READ_TOKEN= +NEXT_PUBLIC_MEDIA_BASE_URL=https://media.epyc.in +``` + +The public website must default to published-only behavior if any CMS mode +configuration is absent or invalid. + +### Preview website + +```text +CMS_PROVIDER=payload +CMS_MODE=draft +DEPLOYMENT_ROLE=preview +PAYLOAD_URL=https://cms.epyc.in +PAYLOAD_PREVIEW_TOKEN= +NEXT_PUBLIC_MEDIA_BASE_URL=https://media.epyc.in +``` + +Draft reads must require both `CMS_MODE=draft` and +`DEPLOYMENT_ROLE=preview`. A single accidental variable must not expose drafts +on the production domain. + +### Existing application D1 + +The website's current `DB` binding remains exclusively responsible for contact +submissions, workshop submissions, chatbot state, and related application +data. Payload must use a separate `PAYLOAD_DB` database. Payload migrations, +restores, and development schema operations must never target the application +database. + +## 4. Current CMS Surface Area + +The following website behavior currently depends on Strapi and must be moved: + +- `/blog`; +- `/blog/[slug]`; +- `/projects`; +- `/gallery`; +- `/gallery/[slug]`; +- blog entries in the sitemap; +- CMS-backed Markdown representations; +- blog metadata, Open Graph data, JSON-LD, and related posts; +- project sorting, filtering, redirect links, and case-study links; +- gallery images, video URLs, related items, and detail metadata; +- media URL resolution and inline rich-text images; +- published-versus-draft reads; and +- CMS cache invalidation. + +The following are explicitly out of scope and must remain unchanged: + +- static homepage featured-project data in `data/projects.ts`; +- hand-authored case studies under `app/(my-app)/case-study/`; +- contact and workshop storage in D1; +- contact webhook queues and workers; +- chatbot application data; +- static site content outside the CMS-backed routes; and +- redesigning CMS content or changing public URL structure. + +## 5. Payload Collections + +The new Payload schema must match the current Strapi content contract, including +fields added after the older Payload implementation was removed. The old +Payload collections in Git history are useful references but are not the +authoritative schema. + +### Users + +- Payload authentication collection. +- Email and optional display name. +- Editor and administrator roles if role separation is needed. +- Secure cookies and normal login throttling. +- Existing Strapi administrator password hashes must not be migrated. Invite or + recreate users and require new passwords. + +### Media + +- R2-backed upload collection. +- Image MIME types initially; add other types only when required. +- Alt text. +- Filename, width, height, MIME type, file size, and public URL/key metadata. +- Focal point only if all consumers and image processing support it. +- Generated sizes should preserve the dimensions consumed by blog/project + cards, or consumers should reliably fall back to the original image. +- Media files must never rely on Worker or container local disk in production. + +### Authors + +- `name`: required text. +- `slug`: required, unique, indexed slug. +- `authorImage`: optional relationship to Media. +- Preserve a legacy Strapi document identifier for migration auditability. + +Do not restore the obsolete `bio` field unless current live Strapi data or a +confirmed editorial requirement needs it. + +### Blogs + +- `title`: required text. +- `slug`: required, unique, indexed slug. +- `publishedDate`: optional date/time used for editorial display and sorting. +- `coverImage`: required relationship to Media for published documents. +- `coverImageAlt`: optional text. +- `author`: required relationship to Authors for published documents. +- `readTime`: optional text. +- `content`: HTML-capable field. +- `metaTitle`: optional text. +- `metaDescription`: optional textarea. +- Legacy Strapi document identifier. +- Payload drafts, versions, autosave, and scheduled publishing. + +Keep the existing CKEditor-produced HTML representation during migration. +Converting all documents to Lexical in the same cutover would combine CMS +migration and content-format migration. A later project can convert rich text +after HTML parity is established. + +### Projects + +- `title`: required text. +- `slug`: required, unique, indexed slug. +- `thumbnail`: required relationship to Media for published documents. +- `thumbnailAlt`: optional text. +- `type`: required multi-select using the existing service/technology values. +- `industry`: required select using the current industry slugs. +- `platform`: required select using the current platform slugs. +- `redirectLink`: required URL/text field with URL validation. +- `caseStudyPath`: optional internal-path field. +- `featured`: boolean, default false, indexed. +- Legacy Strapi document identifier. +- Payload drafts and versions. + +The website compatibility layer may temporarily convert the `type` array to +the existing comma-separated display string. Industry and platform slugs must +remain unchanged because website filtering depends on them. + +Separate industry and platform collections are unnecessary unless editors need +to create values without a code/schema change. Controlled selects are simpler +and preserve the small, fixed vocabulary currently used by the website. + +### Gallery + +- `title`: required text. +- `slug`: required, unique, indexed slug. +- `videoUrl`: optional URL/text field. +- `image`: optional relationship to Media. +- `imageAlt`: optional text. +- `content`: HTML-capable field. +- `designers`: either the current comma-separated text representation or an + array normalized behind the website adapter. +- `externalUrl`: optional URL. +- `year`: optional text. +- Legacy Strapi document identifier. +- Payload drafts and versions. + +Collection validation must require at least one of `image` or `videoUrl`. +Published documents must satisfy all fields currently required by the website. +Incomplete drafts may be allowed so editors can save work in progress. + +## 6. Access Control + +Payload access rules must enforce: + +- anonymous or public website reads return published documents only; +- the public website token, if used, can read published content and media only; +- the preview website token can read drafts but cannot create, update, or + delete content; +- editors can create and edit drafts and publish according to their role; +- migration credentials are temporary and removed after cutover; +- no anonymous collection writes; +- media writes require an authenticated editor or migration process; and +- REST/GraphQL depth, pagination, and query complexity are bounded. + +The preview website must additionally be protected by Cloudflare Access. It +must emit `noindex, nofollow`, an `X-Robots-Tag` equivalent, no public sitemap, +and no shared caching of draft responses. + +## 7. Website Abstraction Before Cutover + +The website must stop importing Strapi-shaped objects directly into routes. +Introduce CMS-neutral domain models: + +```text +Blog +Author +Project +GalleryItem +Media +``` + +Expose provider operations such as: + +```text +listBlogs +getBlogBySlug +listProjects +listGalleryItems +getGalleryItemBySlug +listBlogSlugsForSitemap +``` + +Implement: + +```text +StrapiProvider -> current behavior +PayloadProvider -> target behavior +``` + +All pages, metadata functions, sitemap generation, Markdown builders, and +normalizers must consume only the neutral interface. Provider selection remains +an environment setting until the rollback window closes. + +The provider must support an explicit content state: + +```text +published -> last published Payload version only +draft -> latest available draft/version +``` + +Production must fail closed to `published`. + +## 8. Caching and Revalidation + +Keep the existing 60-second published-content ISR during and immediately after +migration. Add signed Payload webhooks for faster updates. + +### Draft save + +Revalidate only the preview website: + +- the affected detail route; +- its collection index; +- its preview Markdown representation; and +- other preview pages that directly surface the changed item. + +### Publish or unpublish + +Revalidate both preview and production: + +- the affected detail route; +- its collection index; +- the sitemap where applicable; +- its Markdown representation; and +- related-content pages whose selection may change. + +### Delete + +Revalidate both sites and all associated index/sitemap/related paths. + +### Webhook security + +- Sign requests with HMAC or use a strong shared secret. +- Keep the secret server-side. +- Permit only known collection and route mappings. +- Never accept an arbitrary path supplied by the caller. +- Make retries idempotent. +- Log event ID, collection, document ID, action, result, and duration without + logging credentials or document bodies. +- Retain ISR as a recovery path for missed webhooks. + +## 9. Preview Workflow + +The editorial workflow is: + +1. An editor creates or changes a Payload document. +2. The editor saves a draft. +3. Payload invalidates the relevant preview-site paths. +4. The protected preview site requests `draft=true` and displays the latest + draft. +5. The public website continues displaying the last published version. +6. The editor reviews the preview site or Payload Live Preview. +7. The editor publishes the same document. +8. Payload invalidates both preview and public paths. +9. The public website begins displaying the published version. +10. Later draft edits again remain preview-only until the next publish. + +Editors do not copy content between environments and do not write a post twice. + +## 10. Authoritative Strapi Export + +The current CSV files are historical seed inputs, not the migration source of +truth. They contain approximately 3 authors, 27 blogs, 90 projects, 83 gallery +items, 11 industries, and 2 platforms, but they can omit later editorial +changes such as `caseStudyPath`, author images, updated media, and changed +drafts. + +Export from the live Strapi system: + +- published documents; +- the latest draft state where one exists; +- document IDs and slugs; +- created, updated, editorial publication, and system publication timestamps; +- all relations; +- media metadata and every referenced R2 object; +- inline images referenced by rich-text HTML; +- alt text and dimensions; +- project links and case-study paths; and +- draft/published status. + +Generate a machine-readable manifest containing: + +- source type; +- source document ID; +- slug; +- destination collection; +- content hash; +- relationship identifiers; +- media URLs/keys and checksums; +- source status; and +- migration result. + +Historical Strapi revision history is not required for initial parity unless +the business explicitly requires it. Migrate the current published version and +latest draft. Payload begins its own version history after migration. + +## 11. Idempotent Payload Import + +Import in dependency order: + +1. Media. +2. Authors. +3. Project vocabulary values, if modeled as collections. +4. Blogs. +5. Projects. +6. Gallery items. +7. Draft/published state. + +Rules: + +- Upsert using `legacyStrapiDocumentId`, with slug as a secondary check. +- Never rely on Strapi numeric IDs becoming Payload IDs. +- Preserve slugs exactly. +- Preserve publication dates separately from migration timestamps. +- Resolve relationships only after their dependencies exist. +- Store import checkpoints so a failed run can resume. +- Do not duplicate media whose checksum and intended R2 key already match. +- Do not delete Strapi media. +- Parse and rewrite HTML with an HTML parser, not broad regular-expression + replacement. +- Validate every inline image after rewriting. +- Make repeated imports result in updates, not duplicate documents. +- Produce an error report and do not silently skip invalid documents. + +If a document has both a published version and a newer draft, import the +published version first and then create the latest draft so production and +preview show the correct independent states. + +## 12. Media Strategy + +Use R2 for all actual files and D1 only for Payload media records. + +Preferred migration behavior: + +1. Inventory every Strapi media record and inline media URL. +2. Resolve the underlying R2 object. +3. Compute or retrieve a checksum. +4. Reuse the object when ownership and naming are safe, otherwise copy it to a + Payload-controlled key/prefix. +5. Create the Payload media record. +6. Preserve alt text, MIME type, dimensions, and filename. +7. Rewrite document references through the destination media mapping. +8. Verify the public URL, content type, content length, and image dimensions. + +Keep existing media URLs reachable through the rollback window and preferably +longer, because indexed pages, social previews, cached HTML, and external links +may still reference them. + +The existing media-domain abstraction should remain during migration. Changing +the CMS and public media hostname simultaneously is unnecessary risk. + +## 13. Shadow Comparison + +Before Payload serves users, compare it against Strapi while Strapi remains the +response source. + +For each provider operation, compare normalized results in a test job or +sampled shadow read: + +- document counts; +- complete slug sets; +- title and body hashes; +- published/draft state; +- dates and ordering; +- author relationships; +- project industry, platform, type, featured state, and links; +- gallery kind, designers, links, and year; +- media presence, URLs, dimensions, MIME type, and checksums; +- inline HTML media references; +- related-content selection; +- sitemap entries; +- metadata and structured data; and +- Markdown output. + +Differences must be categorized as expected transformations or migration +defects. The cutover cannot proceed with unexplained differences. + +## 14. Verification Matrix + +### Automated checks + +- TypeScript typecheck. +- Unit tests for both providers. +- Provider contract tests. +- Schema validation tests. +- Import idempotency tests. +- Published-versus-draft access tests. +- Webhook signature and route allow-list tests. +- HTML rewrite tests including `src`, `srcset`, absolute URLs, and inline media. +- Sitemap and Markdown route tests. +- Full Next.js build using the installed Next.js documentation and conventions. +- OpenNext/Cloudflare build and preview. +- Payload admin/API build. +- D1 migration status and drift checks. + +### Route checks + +Test every migrated slug, not only representative examples: + +- HTTP status; +- canonical URL; +- title and description; +- Open Graph image; +- JSON-LD; +- visible content; +- images and video; +- internal and external links; +- related items; +- published visibility on the main domain; +- latest-draft visibility on preview; and +- Markdown representation where supported. + +### Acceptance requirements + +- No current published URL becomes a 404. +- No slug changes without an approved permanent redirect. +- No draft appears on the public domain. +- Preview shows the latest draft while production retains the last published + version. +- Every media object loads successfully. +- Blog HTML is semantically and visually equivalent. +- Project ordering and filters are unchanged. +- Sitemap entries are equivalent. +- SEO metadata and structured data remain equivalent. +- Saving a draft invalidates preview only. +- Publishing invalidates both websites. +- CMS failures are observable and do not silently look like an empty site. +- Contact, workshop, chatbot, and queue behavior is unchanged. + +## 15. D1 Migration Discipline + +There is no persistent CMS staging database, so schema changes require a strict +production process. + +For each Payload schema release: + +1. Change the schema locally. +2. Generate a committed migration. +3. Review the generated SQL and Payload configuration diff. +4. Apply the migration to a temporary/local D1 database populated with a + representative schema/data export. +5. Run Payload and the integration test suite against that database. +6. Record the production D1 Time Travel bookmark and create a longer-lived + export to R2 when appropriate. +7. Enter a short CMS maintenance window if the schema and running application + versions are not backward compatible. +8. Apply the production migration. +9. Deploy the matching Payload version. +10. Run admin, API, preview, publish, and public smoke tests. +11. Restore through D1 Time Travel and roll back code if the release fails. + +Production schema push must be disabled. The Payload Worker must not mutate the +schema automatically on startup. + +Large updates and imports must be batched. Rich-text media must remain external +R2 references rather than base64 data in D1. + +## 16. Implementation Phases + +### Phase A: Foundation + +- Create the Payload CMS repository/service. +- Pin compatible Payload, Next.js, OpenNext, and Cloudflare package versions. +- Configure the dedicated D1 binding. +- Configure R2 storage. +- Add Users and Media. +- Add health/readiness checks and structured logging. +- Establish the migration workflow and production backup procedure. + +Exit gate: Payload admin, authentication, D1, R2 upload/read/delete, and a +production-like Cloudflare preview all work. + +### Phase B: Schema parity + +- Implement Authors, Blogs, Projects, and Gallery. +- Add drafts, versions, scheduled publishing, validation, indexes, and access + rules. +- Configure preview URLs and allowed origins. +- Generate and commit the initial D1 schema migration. + +Exit gate: schema contract tests cover every field and state used by the +website. + +### Phase C: Website provider abstraction + +- Add neutral domain types and CMS provider interface. +- Move Strapi REST details behind `StrapiProvider`. +- Update pages, metadata, sitemap, Markdown, and normalizers to use the + interface without changing output. +- Add provider contract tests. + +Exit gate: the website still serves Strapi with no intentional output changes. + +### Phase D: Export/import tooling + +- Build the live Strapi exporter and manifest. +- Build resumable, idempotent Payload importers. +- Implement media mapping and HTML rewriting. +- Import into a temporary/local validation database repeatedly. +- Produce parity reports. + +Exit gate: repeated imports are clean and all source documents/media are +accounted for. + +### Phase E: Payload provider and preview + +- Implement `PayloadProvider`. +- Add published and draft modes with fail-closed production behavior. +- Protect the preview domain with Cloudflare Access. +- Add noindex and cache protections. +- Implement Payload preview/live-preview behavior. +- Add signed revalidation hooks and endpoints. + +Exit gate: save-draft and publish workflows behave correctly end to end. + +### Phase F: Shadow validation + +- Run Strapi and Payload comparisons. +- Crawl all dynamic routes through both providers. +- Run metadata, sitemap, Markdown, media, link, and screenshot comparisons. +- Resolve every unexplained mismatch. +- Rehearse the production migration and rollback procedure. + +Exit gate: all acceptance requirements pass and rollback has been rehearsed. + +### Phase G: Production cutover + +- Announce a short editorial freeze. +- Back up Strapi database and media metadata. +- Record D1 recovery state. +- Run the final Strapi export. +- Run the final idempotent Payload import. +- Validate counts, hashes, relations, status, media, and all public slugs. +- Make Strapi read-only for editors. +- Switch the protected preview website to Payload draft mode. +- Verify drafts and published versions independently. +- Switch the public website to Payload published mode. +- Purge/revalidate affected paths. +- Run the production smoke suite and monitor errors, latency, and 404s. + +Exit gate: preview and production pass all critical checks with Payload as the +active provider. + +### Phase H: Rollback window and cleanup + +- Keep Strapi online and read-only for at least two weeks or two normal + editorial cycles. +- Keep the provider switch available. +- Monitor CMS/API errors, webhook failures, D1 load, media failures, SEO crawl + errors, and unexpected 404s. +- Train editors and document recovery procedures. +- After acceptance, remove Strapi code, secrets, deployment configuration, and + shadow comparison. +- Archive the Strapi repository and retain backups according to the agreed + retention policy. + +## 17. Cutover Rollback + +Rollback must remain possible without reversing Payload writes. + +If the Payload cutover fails: + +1. Set the public and preview websites back to `CMS_PROVIDER=strapi`. +2. Redeploy or roll back the website Worker versions. +3. Revalidate/purge CMS-backed routes. +4. Re-enable Strapi editorial access if necessary. +5. Keep Payload and its imported data intact for diagnosis. +6. Restore Payload D1 only if a Payload schema/data failure requires it. + +Strapi database and media must not be deleted or modified destructively before +the rollback window ends. + +## 18. Observability + +Track at minimum: + +- Payload API response status and latency; +- D1 errors, query latency, overload events, rows read, and rows written; +- R2 upload/read/delete failures; +- webhook attempts, failures, retries, and processing time; +- published and draft document counts; +- website CMS fetch failures; +- CMS-backed route 404s; +- sitemap entry count; +- missing or broken media; and +- preview authorization failures. + +The current behavior of returning an empty CMS list on any upstream failure +must be tightened. Build-time absence may have an explicit fallback, but a +production CMS outage must be logged and surfaced to monitoring rather than +silently rendering empty indexes as a successful response. + +## 19. Decisions Locked by This Plan + +- One Payload CMS, not separate staging and production CMS instances. +- One dedicated production Payload D1 database. +- R2 for media. +- Payload deployed on Cloudflare Workers/OpenNext, not Containers. +- Preview and public websites read the same database. +- Preview reads latest drafts; production reads published versions only. +- Editors write content once. +- Preview is access-controlled and non-indexable. +- The existing application D1 is not reused for Payload. +- HTML rich text is preserved during the initial migration. +- Public routes and slugs do not change. +- Strapi remains available as a rollback source during the acceptance window. +- Schema changes use reviewed migrations, never production schema push. + +## 20. Definition of Done + +The migration is complete when: + +- Payload is the only active CMS provider for preview and production; +- editors can save once, review drafts on preview, and publish to the main + domain; +- every migrated document, relation, status, and media object is accounted for; +- all automated and route-level acceptance requirements pass; +- production has completed the rollback observation window without unresolved + CMS regressions; +- Strapi-specific website code and secrets have been removed; +- Strapi has been archived with recoverable backups; and +- operational, publishing, migration, backup, and recovery documentation has + been handed to the team. diff --git a/lib/blogs/normalise.ts b/lib/blogs/normalise.ts index 0f3bb23..ddfb36b 100644 --- a/lib/blogs/normalise.ts +++ b/lib/blogs/normalise.ts @@ -1,4 +1,4 @@ -import type { StrapiBlog, StrapiMedia } from '../strapi/types' +import type { Blog, Media } from '../cms' export type CoverSize = 'card' | 'banner' @@ -30,13 +30,13 @@ const DATE_FMT: Intl.DateTimeFormatOptions = { year: 'numeric', } -function pickImageUrl(media: StrapiMedia, size: CoverSize): { url: string; width: number; height: number } { +function pickImageUrl(media: Media, size: CoverSize): { url: string; width: number; height: number } { const fmt = size === 'banner' ? media.formats?.large : media.formats?.large if (fmt?.url) return fmt return { url: media.url, width: media.width, height: media.height } } -export function normalise(blog: StrapiBlog, size: CoverSize = 'card'): NormalisedBlog { +export function normalise(blog: Blog, size: CoverSize = 'card'): NormalisedBlog { // Strapi returns `null` for an unset media relation even though the type // says otherwise — guard so a cover-less post doesn't crash the render. const picked = blog.coverImage ? pickImageUrl(blog.coverImage, size) : null diff --git a/lib/cms/config.test.ts b/lib/cms/config.test.ts new file mode 100644 index 0000000..b92ceb7 --- /dev/null +++ b/lib/cms/config.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from 'vitest' +import { getCMSProviderName, getContentState, isPreviewDeployment } from './config' + +describe('CMS configuration', () => { + it('fails closed to published content', () => { + expect(getContentState({ CMS_MODE: 'draft', DEPLOYMENT_ROLE: 'production' })).toBe('published') + expect(getContentState({ CMS_MODE: 'published', DEPLOYMENT_ROLE: 'preview' })).toBe('published') + expect(getContentState({})).toBe('published') + }) + + it('allows drafts only on an explicitly configured preview deployment', () => { + expect(getContentState({ CMS_MODE: 'draft', DEPLOYMENT_ROLE: 'preview' })).toBe('draft') + expect(isPreviewDeployment({ DEPLOYMENT_ROLE: 'preview' })).toBe(true) + }) + + it('keeps Strapi as the rollback default', () => { + expect(getCMSProviderName({})).toBe('strapi') + expect(getCMSProviderName({ CMS_PROVIDER: 'payload' })).toBe('payload') + expect(getCMSProviderName({ CMS_PROVIDER: 'unknown' })).toBe('strapi') + }) +}) diff --git a/lib/cms/config.ts b/lib/cms/config.ts new file mode 100644 index 0000000..6a22091 --- /dev/null +++ b/lib/cms/config.ts @@ -0,0 +1,18 @@ +import type { ContentState } from './types' + +export type CMSProviderName = 'strapi' | 'payload' + +type Environment = Record + +export function getCMSProviderName(env: Environment = process.env): CMSProviderName { + return env.CMS_PROVIDER === 'payload' ? 'payload' : 'strapi' +} + +/** Draft access needs two explicit preview flags. Any invalid configuration fails closed. */ +export function getContentState(env: Environment = process.env): ContentState { + return env.CMS_MODE === 'draft' && env.DEPLOYMENT_ROLE === 'preview' ? 'draft' : 'published' +} + +export function isPreviewDeployment(env: Environment = process.env): boolean { + return env.DEPLOYMENT_ROLE === 'preview' +} diff --git a/lib/cms/index.ts b/lib/cms/index.ts new file mode 100644 index 0000000..d2247c3 --- /dev/null +++ b/lib/cms/index.ts @@ -0,0 +1,16 @@ +import { getCMSProviderName, getContentState } from './config' +import { PayloadProvider } from './payload-provider' +import { StrapiProvider } from './strapi-provider' +import type { CMSProvider } from './types' + +let provider: CMSProvider | undefined + +export function getCMS(): CMSProvider { + if (provider) return provider + provider = getCMSProviderName() === 'payload' + ? new PayloadProvider({ draft: getContentState() === 'draft' }) + : new StrapiProvider() + return provider +} + +export type { Author, Blog, CMSProvider, ContentState, GalleryItem, Media, Project } from './types' diff --git a/lib/cms/payload-provider.test.ts b/lib/cms/payload-provider.test.ts new file mode 100644 index 0000000..7c04492 --- /dev/null +++ b/lib/cms/payload-provider.test.ts @@ -0,0 +1,127 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { PayloadProvider } from './payload-provider' + +type Call = { url: URL; init: RequestInit } + +function stubFetch(docs: unknown[], status = 200) { + const calls: Call[] = [] + vi.stubGlobal('fetch', (input: URL | string, init: RequestInit = {}) => { + calls.push({ url: new URL(String(input)), init }) + return Promise.resolve( + new Response(JSON.stringify({ docs }), { status, headers: { 'content-type': 'application/json' } }), + ) + }) + return calls +} + +const provider = (draft = false) => new PayloadProvider({ baseUrl: 'https://cms.test', token: 'tok', draft }) + +const mediaDoc = { + id: 7, + url: '/media/cover.png', + width: 2400, + height: 1350, + alt: 'A cover', + sizes: { + thumbnail: { url: '/media/cover-400.png', width: 400, height: 225 }, + card: { url: '/media/cover-1080.png', width: 1080, height: 608 }, + banner: { url: '/media/cover-1600.png', width: 1600, height: 900 }, + }, +} + +afterEach(() => vi.unstubAllGlobals()) + +describe('PayloadProvider media mapping', () => { + it('aliases Payload sizes onto the format names the normalisers read', async () => { + stubFetch([{ id: 1, title: 'Post', slug: 'post', updatedAt: '2026-01-01', coverImage: mediaDoc }]) + const [blog] = await provider().listBlogs() + + // `lib/blogs/normalise.ts` and `lib/projects/normalise.ts` read + // `formats.large`; without the alias every image serves at full size. + expect(blog.coverImage?.formats?.large).toEqual({ url: '/media/cover-1080.png', width: 1080, height: 608 }) + expect(blog.coverImage?.formats?.banner?.width).toBe(1600) + expect(blog.coverImage?.alternativeText).toBe('A cover') + expect(blog.coverImage?.id).toBe('7') + }) + + it('drops a size that is missing dimensions instead of emitting a partial format', async () => { + stubFetch([ + { + id: 1, + title: 'Post', + slug: 'post', + updatedAt: '2026-01-01', + coverImage: { ...mediaDoc, sizes: { card: { url: '/media/cover-1080.png' } } }, + }, + ]) + const [blog] = await provider().listBlogs() + expect(blog.coverImage?.formats?.large).toBeUndefined() + expect(blog.coverImage?.url).toBe('/media/cover.png') + }) + + it('treats an unpopulated relation and a file-less upload as absent', async () => { + stubFetch([ + { id: 1, title: 'Unpopulated', slug: 'a', updatedAt: '2026-01-01', coverImage: 7, author: 3 }, + { id: 2, title: 'No file', slug: 'b', updatedAt: '2026-01-01', coverImage: { id: 8, url: null } }, + ]) + const [unpopulated, noFile] = await provider().listBlogs() + expect(unpopulated.coverImage).toBeNull() + expect(unpopulated.author).toBeNull() + expect(noFile.coverImage).toBeNull() + }) +}) + +describe('PayloadProvider requests', () => { + it('fetches a single blog by slug in one request', async () => { + const calls = stubFetch([{ id: 1, title: 'Post', slug: 'post', updatedAt: '2026-01-01' }]) + const blog = await provider().getBlogBySlug('post') + + expect(calls).toHaveLength(1) + expect(calls[0].url.searchParams.get('where[slug][equals]')).toBe('post') + expect(calls[0].url.searchParams.get('limit')).toBe('1') + expect(blog?.slug).toBe('post') + }) + + it('reads published content with ISR and drafts with no caching', async () => { + const published = stubFetch([]) + await provider().listBlogs() + expect(published[0].url.searchParams.get('draft')).toBe('false') + expect((published[0].init as { next?: { revalidate?: number } }).next?.revalidate).toBe(60) + expect(published[0].init.cache).toBeUndefined() + + vi.unstubAllGlobals() + const drafts = stubFetch([]) + await provider(true).listBlogs() + expect(drafts[0].url.searchParams.get('draft')).toBe('true') + expect(drafts[0].init.cache).toBe('no-store') + }) + + it('excludes a slug for related-item lists', async () => { + const calls = stubFetch([]) + await provider().listBlogs({ excludeSlug: 'post', limit: 3 }) + expect(calls[0].url.searchParams.get('where[slug][not_equals]')).toBe('post') + expect(calls[0].url.searchParams.get('limit')).toBe('3') + }) + + it('throws on an upstream failure rather than reporting an empty collection', async () => { + stubFetch([], 502) + await expect(provider().listBlogs()).rejects.toThrow('Payload 502: blogs') + }) + + it('requires a base URL', async () => { + await expect(new PayloadProvider({ baseUrl: '' }).listBlogs()).rejects.toThrow('PAYLOAD_URL is required') + }) +}) + +describe('PayloadProvider field shapes', () => { + it('keeps project type as an array and defaults gallery designers', async () => { + stubFetch([{ id: 1, title: 'P', slug: 'p', type: ['WEBFLOW', 'SEO'], featured: true, publishedAt: '2026-01-01' }]) + const [project] = await provider().listProjects() + expect(project.type).toEqual(['WEBFLOW', 'SEO']) + + vi.unstubAllGlobals() + stubFetch([{ id: 2, title: 'G', slug: 'g' }]) + const [item] = await provider().listGalleryItems() + expect(item.designers).toEqual([]) + }) +}) diff --git a/lib/cms/payload-provider.ts b/lib/cms/payload-provider.ts new file mode 100644 index 0000000..62e83ec --- /dev/null +++ b/lib/cms/payload-provider.ts @@ -0,0 +1,140 @@ +import type { Author, Blog, CMSProvider, GalleryItem, ListOptions, Media, MediaFormat, Project } from './types' + +type PayloadList = { docs: T[]; hasNextPage?: boolean; nextPage?: number | null } +type PayloadRelation = T | string | number | null +type PayloadSize = { url?: string | null; width?: number | null; height?: number | null } +type PayloadMedia = Omit & { + id: string | number + url?: string | null + alt?: string | null + sizes?: Record +} +type PayloadAuthor = Omit & { id: string | number; authorImage?: PayloadRelation } +type PayloadBlog = Omit & { + id: string | number + coverImage?: PayloadRelation + author?: PayloadRelation + _status?: 'draft' | 'published' + publishedAt?: string | null +} +type PayloadProject = Omit & { id: string | number; thumbnail?: PayloadRelation } +type PayloadGallery = Omit & { id: string | number; image?: PayloadRelation } + +function populated(value: PayloadRelation | undefined): value is T { + return typeof value === 'object' && value !== null +} + +function size(value?: PayloadSize): MediaFormat | undefined { + if (!value?.url || !value.width || !value.height) return undefined + return { url: value.url, width: value.width, height: value.height } +} + +function mapMedia(value?: PayloadRelation): Media | null { + if (!populated(value) || !value.url) return null + // Consumers (`lib/blogs/normalise.ts`, `lib/projects/normalise.ts`) ask for + // `formats.large`, which is Strapi's name for the resized copy. Payload calls + // its equivalent `sizes.card`, so alias it — without this every image falls + // back to the full-size original. + const formats: Record = { + thumbnail: size(value.sizes?.thumbnail), + card: size(value.sizes?.card), + large: size(value.sizes?.card), + banner: size(value.sizes?.banner), + } + return { + ...value, + id: String(value.id), + url: value.url, + alternativeText: value.alternativeText ?? value.alt ?? null, + formats, + } +} + +function mapBlog(item: PayloadBlog): Blog { + return { + ...item, + id: String(item.id), + publishedAt: item.publishedAt ?? item.publishedDate ?? item.updatedAt, + coverImage: mapMedia(item.coverImage), + author: populated(item.author) + ? { ...item.author, id: String(item.author.id), authorImage: mapMedia(item.author.authorImage) } + : null, + } +} + +function mapGalleryItem(item: PayloadGallery): GalleryItem { + return { ...item, id: String(item.id), designers: item.designers ?? [], image: mapMedia(item.image) } +} + +export class PayloadProvider implements CMSProvider { + private readonly base: string + private readonly token?: string + private readonly draft: boolean + + constructor(options: { baseUrl?: string; token?: string; draft?: boolean } = {}) { + this.base = (options.baseUrl ?? process.env.PAYLOAD_URL ?? '').replace(/\/$/, '') + this.draft = options.draft ?? false + this.token = options.token ?? (this.draft ? process.env.PAYLOAD_PREVIEW_TOKEN : process.env.PAYLOAD_READ_TOKEN) + } + + private async list(collection: string, params: Record = {}): Promise { + if (!this.base) throw new Error('Payload: PAYLOAD_URL is required when CMS_PROVIDER=payload') + const url = new URL(`/api/${collection}`, this.base) + Object.entries({ depth: '2', limit: '100', draft: String(this.draft), ...params }).forEach(([key, value]) => + url.searchParams.set(key, value), + ) + const response = await fetch(url, { + headers: this.token ? { Authorization: `users API-Key ${this.token}` } : {}, + ...(this.draft ? { cache: 'no-store' as const } : { next: { revalidate: 60, tags: [`cms:${collection}`] } }), + }) + if (!response.ok) throw new Error(`Payload ${response.status}: ${collection}`) + const body = (await response.json()) as PayloadList + return body.docs + } + + private async one(collection: string, slug: string): Promise { + const docs = await this.list(collection, { 'where[slug][equals]': slug, limit: '1' }) + return docs[0] ?? null + } + + async listBlogs(options: ListOptions = {}): Promise { + const docs = await this.list('blogs', { + sort: '-publishedDate', + limit: String(options.limit ?? 100), + ...(options.excludeSlug ? { 'where[slug][not_equals]': options.excludeSlug } : {}), + }) + return docs.map(mapBlog) + } + + async getBlogBySlug(slug: string) { + const item = await this.one('blogs', slug) + return item ? mapBlog(item) : null + } + + async listProjects(options: ListOptions = {}): Promise { + const docs = await this.list('projects', { sort: '-featured,-publishedAt', limit: String(options.limit ?? 200) }) + return docs.map((item) => ({ ...item, id: String(item.id), type: Array.isArray(item.type) ? item.type : [], thumbnail: mapMedia(item.thumbnail) })) + } + + async listGalleryItems(options: ListOptions = {}): Promise { + const docs = await this.list('gallery', { + // /gallery renders in Strapi's numeric id order, preserved on import as + // legacyStrapiId. Payload's default createdAt order would be the import + // order, which is arbitrary. + sort: 'legacyStrapiId', + limit: String(options.limit ?? 500), + ...(options.excludeSlug ? { 'where[slug][not_equals]': options.excludeSlug } : {}), + }) + return docs.map(mapGalleryItem) + } + + async getGalleryItemBySlug(slug: string) { + const item = await this.one('gallery', slug) + return item ? mapGalleryItem(item) : null + } + + async listBlogSlugsForSitemap() { + const blogs = await this.listBlogs({ limit: 1000 }) + return blogs.map(({ slug, publishedAt }) => ({ slug, publishedAt })) + } +} diff --git a/lib/cms/revalidation.test.ts b/lib/cms/revalidation.test.ts new file mode 100644 index 0000000..aa99e64 --- /dev/null +++ b/lib/cms/revalidation.test.ts @@ -0,0 +1,40 @@ +import { createHmac } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { parseRevalidationEvent, pathsForEvent, tagsForEvent, verifyWebhookSignature } from './revalidation' + +describe('CMS revalidation', () => { + it('maps only known collections to application-owned paths', () => { + expect(pathsForEvent({ eventId: '1', collection: 'blogs', action: 'publish', slug: 'hello-world' })).toEqual([ + '/blog', '/blog/hello-world', '/sitemap.xml', + ]) + expect(pathsForEvent({ eventId: '2', collection: 'projects', action: 'publish', slug: 'ignored' })).toEqual(['/projects']) + }) + + it('purges the provider fetch tag alongside the paths', () => { + expect(tagsForEvent({ eventId: '1', collection: 'gallery', action: 'publish' })).toEqual(['cms:gallery']) + }) + + it('accepts the punctuation that appears in real slugs', () => { + expect(parseRevalidationEvent({ eventId: '1', collection: 'projects', action: 'publish', slug: 'breathewellbeing.in' })?.slug).toBe('breathewellbeing.in') + expect(parseRevalidationEvent({ eventId: '1', collection: 'gallery', action: 'publish', slug: 'epyc-merchandise-tshirt-concept-design-(green-variant)' })).not.toBeNull() + }) + + it('still refuses anything that could escape the collection route', () => { + for (const slug of ['../admin', 'a/b', 'a..b', 'a b', '%2e%2e', '/etc/passwd', '-leading-hyphen']) { + expect(parseRevalidationEvent({ eventId: '1', collection: 'blogs', action: 'publish', slug })).toBeNull() + } + }) + + it('rejects arbitrary paths and malformed events', () => { + expect(parseRevalidationEvent({ eventId: '1', collection: 'blogs', action: 'publish', slug: '../admin' })).toBeNull() + expect(parseRevalidationEvent({ eventId: '1', collection: 'unknown', action: 'publish' })).toBeNull() + expect(parseRevalidationEvent({ eventId: '1', collection: 'blogs', action: 'execute' })).toBeNull() + }) + + it('verifies HMAC signatures', async () => { + const body = JSON.stringify({ eventId: '1', collection: 'blogs', action: 'publish' }) + const signature = createHmac('sha256', 'secret').update(body).digest('hex') + await expect(verifyWebhookSignature(body, `sha256=${signature}`, 'secret')).resolves.toBe(true) + await expect(verifyWebhookSignature(body, signature.replace(/^./, '0'), 'secret')).resolves.toBe(false) + }) +}) diff --git a/lib/cms/revalidation.ts b/lib/cms/revalidation.ts new file mode 100644 index 0000000..cd2352f --- /dev/null +++ b/lib/cms/revalidation.ts @@ -0,0 +1,61 @@ +export type CMSCollection = 'blogs' | 'projects' | 'gallery' +export type CMSAction = 'draft' | 'publish' | 'unpublish' | 'delete' + +export type RevalidationEvent = { + eventId: string + collection: CMSCollection + action: CMSAction + slug?: string +} + +export function pathsForEvent(event: RevalidationEvent): string[] { + const routes: Record = { + blogs: { index: '/blog', detail: event.slug ? `/blog/${event.slug}` : undefined }, + projects: { index: '/projects' }, + gallery: { index: '/gallery', detail: event.slug ? `/gallery/${event.slug}` : undefined }, + } + const selected = routes[event.collection] + return [...new Set([selected.index, selected.detail, event.collection === 'blogs' ? '/sitemap.xml' : undefined].filter((path): path is string => Boolean(path)))] +} + +/** The Payload provider tags its fetches `cms:`. Purging the path + * alone leaves those cached responses in place, so the page can re-render from + * stale data. */ +export function tagsForEvent(event: RevalidationEvent): string[] { + return [`cms:${event.collection}`] +} + +export function parseRevalidationEvent(value: unknown): RevalidationEvent | null { + if (!value || typeof value !== 'object') return null + const event = value as Record + if (typeof event.eventId !== 'string' || !event.eventId) return null + if (!['blogs', 'projects', 'gallery'].includes(String(event.collection))) return null + if (!['draft', 'publish', 'unpublish', 'delete'].includes(String(event.action))) return null + // Real slugs include dots and parentheses ("breathewellbeing.in", + // "…-(green-variant)"), so the character class cannot be letters and hyphens + // alone. What must stay impossible is escaping the collection's own route: + // no slashes, no traversal, no whitespace, no encoded separators. + if (event.slug !== undefined) { + if (typeof event.slug !== 'string') return null + if (!/^[a-z0-9][a-z0-9._()-]*$/i.test(event.slug)) return null + if (event.slug.includes('..')) return null + } + return event as RevalidationEvent +} + +export async function verifyWebhookSignature(body: string, signature: string | null, secret: string): Promise { + if (!signature || !secret) return false + const expected = await crypto.subtle.sign( + 'HMAC', + await crypto.subtle.importKey('raw', new TextEncoder().encode(secret), { name: 'HMAC', hash: 'SHA-256' }, false, ['sign']), + new TextEncoder().encode(body), + ) + const actual = signature.replace(/^sha256=/, '') + if (!/^[a-f0-9]{64}$/i.test(actual)) return false + const bytes = new Uint8Array(actual.match(/.{2}/g)!.map((byte) => Number.parseInt(byte, 16))) + if (bytes.length !== expected.byteLength) return false + let difference = 0 + const expectedBytes = new Uint8Array(expected) + for (let index = 0; index < bytes.length; index += 1) difference |= bytes[index] ^ expectedBytes[index] + return difference === 0 +} diff --git a/lib/cms/strapi-provider.ts b/lib/cms/strapi-provider.ts new file mode 100644 index 0000000..250b072 --- /dev/null +++ b/lib/cms/strapi-provider.ts @@ -0,0 +1,96 @@ +import { fetchStrapi } from '@/lib/strapi/client' +import type { StrapiBlog, StrapiGalleryItem, StrapiList, StrapiMedia, StrapiProject } from '@/lib/strapi/types' +import type { Blog, CMSProvider, GalleryItem, ListOptions, Media, Project } from './types' + +const BLOG_POPULATE = { + 'populate[coverImage][fields]': 'url,width,height,alternativeText,formats', + 'populate[author][fields]': 'name,slug', +} +const GALLERY_POPULATE = { 'populate[image][fields]': 'url,width,height,alternativeText' } + +function media(value?: StrapiMedia | null): Media | null { + if (!value) return null + return { ...value, id: String(value.id) } +} + +function blog(value: StrapiBlog): Blog { + return { + ...value, + id: value.documentId || String(value.id), + coverImage: media(value.coverImage), + author: value.author ? { ...value.author, id: String(value.author.id) } : null, + } +} + +function project(value: StrapiProject): Project { + return { + ...value, + id: value.documentId || String(value.id), + thumbnail: media(value.thumbnail), + type: (value.type ?? '').split(',').map((item) => item.trim()).filter(Boolean), + } +} + +function gallery(value: StrapiGalleryItem): GalleryItem { + return { + ...value, + id: value.documentId || String(value.id), + image: media(value.image), + designers: (value.designer ?? '').split(',').map((item) => item.trim()).filter(Boolean), + } +} + +export class StrapiProvider implements CMSProvider { + async listBlogs(options: ListOptions = {}): Promise { + const { data } = await fetchStrapi>('/blogs', { + ...BLOG_POPULATE, + ...(options.excludeSlug ? { 'filters[slug][$ne]': options.excludeSlug } : {}), + sort: 'publishedDate:desc', + 'pagination[limit]': String(options.limit ?? 100), + }) + return data.filter((item) => item.slug).map(blog) + } + + async getBlogBySlug(slug: string): Promise { + const { data } = await fetchStrapi>('/blogs', { + 'filters[slug][$eq]': slug, + ...BLOG_POPULATE, + 'pagination[limit]': '1', + }) + return data[0] ? blog(data[0]) : null + } + + async listProjects(options: ListOptions = {}): Promise { + const { data } = await fetchStrapi>('/projects', { + 'populate[thumbnail][fields]': 'url,width,height,alternativeText,formats', + 'populate[industry][fields]': 'title,slug', + 'populate[platform][fields]': 'title,slug', + sort: 'featured:desc,publishedAt:desc', + 'pagination[limit]': String(options.limit ?? 200), + }) + return data.filter((item) => item.slug).map(project) + } + + async listGalleryItems(options: ListOptions = {}): Promise { + const { data } = await fetchStrapi>('/gallery-items', { + ...GALLERY_POPULATE, + ...(options.excludeSlug ? { 'filters[slug][$ne]': options.excludeSlug } : {}), + 'pagination[limit]': String(options.limit ?? 500), + }) + return data.filter((item) => item.slug).map(gallery) + } + + async getGalleryItemBySlug(slug: string): Promise { + const { data } = await fetchStrapi>('/gallery-items', { + 'filters[slug][$eq]': slug, + ...GALLERY_POPULATE, + 'pagination[limit]': '1', + }) + return data[0] ? gallery(data[0]) : null + } + + async listBlogSlugsForSitemap() { + const blogs = await this.listBlogs({ limit: 1000 }) + return blogs.map(({ slug, publishedAt }) => ({ slug, publishedAt })) + } +} diff --git a/lib/cms/types.ts b/lib/cms/types.ts new file mode 100644 index 0000000..0a59c15 --- /dev/null +++ b/lib/cms/types.ts @@ -0,0 +1,74 @@ +export type ContentState = 'published' | 'draft' + +export type MediaFormat = { url: string; width: number; height: number } + +export type Media = { + id: string + url: string + width: number + height: number + alternativeText?: string | null + formats?: Record +} + +export type Author = { + id: string + name: string + slug: string + authorImage?: Media | null +} + +export type Blog = { + id: string + title: string + slug: string + publishedDate: string | null + publishedAt: string + updatedAt: string + coverImage?: Media | null + coverImageAlt?: string | null + author?: Author | null + readTime?: string | null + content?: string | null + metaTitle?: string | null + metaDescription?: string | null +} + +export type Project = { + id: string + title: string + slug: string + publishedAt: string + thumbnail?: Media | null + thumbnailAlt?: string | null + type: string[] + industry?: { title: string; slug: string } | null + platform?: { title: string; slug: string } | null + redirectLink?: string | null + caseStudyPath?: string | null + featured: boolean +} + +export type GalleryItem = { + id: string + title: string + slug: string + image?: Media | null + imageAlt?: string | null + videoUrl?: string | null + content?: string | null + designers?: string[] + externalUrl?: string | null + year?: string | null +} + +export type ListOptions = { limit?: number; excludeSlug?: string } + +export interface CMSProvider { + listBlogs(options?: ListOptions): Promise + getBlogBySlug(slug: string): Promise + listProjects(options?: ListOptions): Promise + listGalleryItems(options?: ListOptions): Promise + getGalleryItemBySlug(slug: string): Promise + listBlogSlugsForSitemap(): Promise>> +} diff --git a/lib/gallery/normalise.ts b/lib/gallery/normalise.ts index 048aedf..89c4d81 100644 --- a/lib/gallery/normalise.ts +++ b/lib/gallery/normalise.ts @@ -1,13 +1,13 @@ -import type { StrapiGalleryItem } from '../strapi/types' +import type { GalleryItem as CMSGalleryItem } from '../cms' import type { GalleryItem } from '../../data/gallery' -function splitDesigners(raw: string | null | undefined): string[] | undefined { - if (!raw) return undefined - const list = raw.split(',').map((s) => s.trim()).filter(Boolean) - return list.length > 0 ? list : undefined +// Providers hand over an array; `data/gallery.ts` wants the field absent rather +// than empty when there are no designers. +function designers(list: string[] | undefined): string[] | undefined { + return list && list.length > 0 ? list : undefined } -export function normaliseGallery(item: StrapiGalleryItem): GalleryItem { +export function normaliseGallery(item: CMSGalleryItem): GalleryItem { const slug = item.slug const videoUrl = item.videoUrl?.trim() const image = item.image ?? null @@ -22,7 +22,7 @@ export function normaliseGallery(item: StrapiGalleryItem): GalleryItem { alt: item.imageAlt ?? image?.alternativeText ?? item.title, title: item.title, description: item.content ?? undefined, - designers: splitDesigners(item.designer), + designers: designers(item.designers), previewLink: item.externalUrl ?? undefined, } } @@ -39,7 +39,7 @@ export function normaliseGallery(item: StrapiGalleryItem): GalleryItem { height: image?.height ?? 1350, title: item.title, description: item.content ?? undefined, - designers: splitDesigners(item.designer), + designers: designers(item.designers), previewLink: item.externalUrl ?? undefined, } } diff --git a/lib/markdown/sources.ts b/lib/markdown/sources.ts index de6c8b3..63504d4 100644 --- a/lib/markdown/sources.ts +++ b/lib/markdown/sources.ts @@ -1,12 +1,11 @@ // Hand-built markdown for the CMS-driven routes. // -// These pages have structured Strapi fields behind them, so building markdown +// These pages have structured CMS fields behind them, so building markdown // from the fields beats converting the rendered HTML — the output carries the // author, date, and tags as data instead of as layout. Every other route falls // back to converting its own rendered HTML (see `app/md/[[...path]]/route.ts`). -import { fetchStrapi } from '@/lib/strapi/client' -import type { StrapiBlog, StrapiGalleryItem, StrapiList, StrapiProject } from '@/lib/strapi/types' +import { getCMS } from '@/lib/cms' import { normalise } from '@/lib/blogs/normalise' import { normaliseGallery } from '@/lib/gallery/normalise' import { normaliseProject } from '@/lib/projects/normalise' @@ -25,18 +24,8 @@ export function frontMatter(fields: Record): string return lines.length > 0 ? `---\n${lines.join('\n')}\n---\n\n` : '' } -const BLOG_FIELDS = { - 'populate[coverImage][fields]': 'url,width,height,alternativeText,formats', - 'populate[author][fields]': 'name,slug', -} - const blogPost: Builder = async ({ slug, origin }) => { - const { data } = await fetchStrapi>('/blogs', { - 'filters[slug][$eq]': slug, - ...BLOG_FIELDS, - 'pagination[limit]': '1', - }) - const raw = data[0] + const raw = await getCMS().getBlogBySlug(slug) if (!raw) return null const blog = normalise(raw, 'banner') @@ -67,12 +56,8 @@ const blogPost: Builder = async ({ slug, origin }) => { } const blogIndex: Builder = async ({ origin }) => { - const { data } = await fetchStrapi>('/blogs', { - ...BLOG_FIELDS, - sort: 'publishedDate:desc', - 'pagination[limit]': '100', - }) - const blogs = data.filter((b) => b.slug).map((b) => normalise(b)) + const data = await getCMS().listBlogs({ limit: 100 }) + const blogs = data.map((b) => normalise(b)) const items = blogs.map((blog) => { const meta = [blog.author, blog.date, blog.readTime].filter(Boolean).join(' · ') @@ -101,14 +86,8 @@ const blogIndex: Builder = async ({ origin }) => { } const projectsIndex: Builder = async ({ origin }) => { - const { data } = await fetchStrapi>('/projects', { - 'populate[thumbnail][fields]': 'url,width,height,alternativeText,formats', - 'populate[industry][fields]': 'title,slug', - 'populate[platform][fields]': 'title,slug', - sort: 'featured:desc,publishedAt:desc', - 'pagination[limit]': '200', - }) - const projects = data.filter((p) => p.slug).map((p) => normaliseProject(p)) + const data = await getCMS().listProjects({ limit: 200 }) + const projects = data.map((p) => normaliseProject(p)) const rows = projects.map((project) => { const link = project.caseStudyPath ? `${origin}${project.caseStudyPath}` : project.redirectLink @@ -139,14 +118,9 @@ const projectsIndex: Builder = async ({ origin }) => { } } -const GALLERY_FIELDS = { 'populate[image][fields]': 'url,width,height,alternativeText' } - const galleryIndex: Builder = async ({ origin }) => { - const { data } = await fetchStrapi>('/gallery-items', { - ...GALLERY_FIELDS, - 'pagination[limit]': '500', - }) - const items = data.filter((item) => item.slug).map((item) => normaliseGallery(item)) + const data = await getCMS().listGalleryItems({ limit: 500 }) + const items = data.map((item) => normaliseGallery(item)) const list = items.map((item) => { const meta = [item.kind, item.designers?.join(', ')].filter(Boolean).join(' · ') @@ -171,12 +145,7 @@ const galleryIndex: Builder = async ({ origin }) => { } const galleryItem: Builder = async ({ slug, origin }) => { - const { data } = await fetchStrapi>('/gallery-items', { - 'filters[slug][$eq]': slug, - ...GALLERY_FIELDS, - 'pagination[limit]': '1', - }) - const raw = data[0] + const raw = await getCMS().getGalleryItemBySlug(slug) if (!raw) return null const item = normaliseGallery(raw) diff --git a/lib/projects/normalise.ts b/lib/projects/normalise.ts index e3ce2d0..531cfe4 100644 --- a/lib/projects/normalise.ts +++ b/lib/projects/normalise.ts @@ -1,4 +1,4 @@ -import type { StrapiProject, StrapiMedia } from '../strapi/types' +import type { Media, Project } from '../cms' export type ProjectIndustry = | 'vc' @@ -36,18 +36,18 @@ export type NormalisedProject = { } | null } -function pickImageUrl(media: StrapiMedia): { url: string; width: number; height: number } { +function pickImageUrl(media: Media): { url: string; width: number; height: number } { const fmt = media.formats?.large if (fmt?.url) return fmt return { url: media.url, width: media.width, height: media.height } } -export function normaliseProject(project: StrapiProject): NormalisedProject { +export function normaliseProject(project: Project): NormalisedProject { // Strapi returns null for unset media/relations in draft mode even though // the types say required — guard every dereference so an incomplete draft // renders with placeholders/fallbacks instead of crashing the page. const picked = project.thumbnail ? pickImageUrl(project.thumbnail) : null - const types = (project.type ?? '').split(',').map((s) => s.trim()).filter(Boolean) + const types = project.type const link = project.redirectLink ?? '' return { @@ -58,7 +58,7 @@ export function normaliseProject(project: StrapiProject): NormalisedProject { industry: (project.industry?.slug ?? 'other') as ProjectIndustry, platform: (project.platform?.slug ?? 'website') as ProjectPlatform, types, - typesDisplay: project.type ?? '', + typesDisplay: project.type.join(', '), featured: Boolean(project.featured), createdAt: project.publishedAt, image: picked diff --git a/lib/strapi/client.ts b/lib/strapi/client.ts index d13ebdd..7dfb3ce 100644 --- a/lib/strapi/client.ts +++ b/lib/strapi/client.ts @@ -2,7 +2,9 @@ const BASE = process.env.STRAPI_URL const TOKEN = process.env.STRAPI_API_TOKEN // When true, fetch Strapi Draft & Publish drafts (`?status=draft`) instead of // published content. Drafts are never edge-cached so editors see live edits. -const PREVIEW = process.env.STRAPI_PREVIEW === 'true' +const PREVIEW = + process.env.DEPLOYMENT_ROLE === 'preview' && + (process.env.CMS_MODE === 'draft' || process.env.STRAPI_PREVIEW === 'true') const EMPTY = { data: [], meta: { pagination: { start: 0, limit: 0, total: 0 } } } diff --git a/middleware.ts b/middleware.ts index 91df238..43b29b2 100644 --- a/middleware.ts +++ b/middleware.ts @@ -30,16 +30,16 @@ export function middleware(request: NextRequest) { // The markdown renderer fetches this app's own HTML to convert it. Negotiating // that request would rewrite it straight back to the renderer. - if (request.headers.has(RENDER_GUARD_HEADER)) return NextResponse.next() + if (request.headers.has(RENDER_GUARD_HEADER)) return protectPreview(NextResponse.next()) - if (request.method !== 'GET' && request.method !== 'HEAD') return NextResponse.next() - if (!isNegotiablePath(pathname)) return NextResponse.next() + if (request.method !== 'GET' && request.method !== 'HEAD') return protectPreview(NextResponse.next()) + if (!isNegotiablePath(pathname)) return protectPreview(NextResponse.next()) if (!prefersMarkdown(request.headers.get('accept'))) { // Same URL, two representations — tell caches the Accept header matters. const response = NextResponse.next() response.headers.set('Vary', 'Accept') - return response + return protectPreview(response) } const target = request.nextUrl.clone() @@ -50,6 +50,13 @@ export function middleware(request: NextRequest) { const response = NextResponse.rewrite(target, { request: { headers: forwarded } }) response.headers.set('Vary', 'Accept') + return protectPreview(response) +} + +function protectPreview(response: NextResponse): NextResponse { + if (process.env.DEPLOYMENT_ROLE !== 'preview') return response + response.headers.set('X-Robots-Tag', 'noindex, nofollow') + response.headers.set('Cache-Control', 'private, no-store, max-age=0') return response } diff --git a/package.json b/package.json index 48e9043..e705f8d 100644 --- a/package.json +++ b/package.json @@ -11,7 +11,8 @@ "preview": "opennextjs-cloudflare build && opennextjs-cloudflare preview", "deploy:staging": "NEXT_PUBLIC_DEPLOY_ENV=staging opennextjs-cloudflare build && wrangler deploy --env staging && wrangler deploy --env staging --config workers/contact-webhook/wrangler.jsonc", "deploy:production": "NEXT_PUBLIC_DEPLOY_ENV=production opennextjs-cloudflare build && wrangler deploy --env production && wrangler deploy --env production --config workers/contact-webhook/wrangler.jsonc", - "cf-typegen": "wrangler types --env-interface CloudflareEnv ./cloudflare-env.d.ts" + "cf-typegen": "wrangler types --env-interface CloudflareEnv ./cloudflare-env.d.ts", + "cms:parity": "tsx scripts/cms-parity.mts" }, "dependencies": { "@hookform/resolvers": "^5.2.2", diff --git a/scripts/cms-parity.ts b/scripts/cms-parity.ts new file mode 100644 index 0000000..efe6917 --- /dev/null +++ b/scripts/cms-parity.ts @@ -0,0 +1,205 @@ +// Compares what the website would render from Strapi against what it would +// render from Payload, through the same provider interface the pages use. +// +// The comparison runs on normalised output — the shape the components actually +// receive — rather than on raw API responses, so a difference here is a +// difference a visitor could see. +// +// Some differences are intended by the migration. Those are declared below and +// reported as "explained"; anything else is a defect and fails the run. +// +// Usage: +// STRAPI_URL=https://cms.epyc.in STRAPI_API_TOKEN=... \ +// PAYLOAD_URL=https://epyc-payload-cms.epyc.workers.dev \ +// pnpm cms:parity + +import { writeFile, mkdir } from 'node:fs/promises' +import { PayloadProvider } from '../lib/cms/payload-provider' +import { StrapiProvider } from '../lib/cms/strapi-provider' +import type { CMSProvider } from '../lib/cms/types' +import { normalise } from '../lib/blogs/normalise' +import { normaliseProject } from '../lib/projects/normalise' +import { normaliseGallery } from '../lib/gallery/normalise' + +const REPORT = 'artifacts/parity-report.json' + +type Diff = { entity: string; key: string; field: string; strapi: unknown; payload: unknown; explained?: string } + +/** Media URLs changed shape twice over: Strapi served a resized derivative at a + * bare path (`/large_x.webp`), Payload records the original at an absolute URL + * (`https://media.epyc.in/x.webp`). Cloudflare resizes either form on request — + * verified against the live zone — so only the underlying file must match. */ +function canonicalMedia(value: unknown): string { + return String(value ?? '') + .replace(/^https?:\/\/[^/]+/, '') + .replace(/(^|\/)(large|medium|small|thumbnail)_/, '$1') +} + +/** Differences the migration intends. Each returns a reason when it recognises + * the pair, so an intended change is never mistaken for a defect — and a defect + * is never waved through as intended. */ +const EXPECTED: Array<(diff: Omit) => string | undefined> = [ + // Strapi served a pre-resized derivative (`/large_x.webp`); Payload records + // the original (`/x.webp`) because resizing happens at Cloudflare's edge. + ({ field, strapi, payload }) => { + if (!/image|src|url|thumbnail|cover/i.test(field)) return undefined + if (!strapi || !payload) return undefined + return canonicalMedia(strapi) === canonicalMedia(payload) + ? 'Original at an absolute URL replaces Strapi derivative; Cloudflare resizes on request' + : undefined + }, + // Image dimensions follow from the same change. + ({ field, strapi, payload }) => + /width|height/i.test(field) && typeof strapi === 'number' && typeof payload === 'number' && payload >= strapi + ? 'Original image is larger than the derivative Strapi served' + : undefined, + // Duplicate service spellings folded on import. + ({ field, strapi, payload }) => { + if (!/types|typesDisplay/i.test(field)) return undefined + const fold = (value: unknown) => + String(value ?? '') + .replace(/UI\/UX|UX-UI/g, 'UI-UX') + .replace(/\bDEV\b/g, 'DEVELOPMENT') + return fold(strapi) === fold(payload) ? 'Duplicate type spellings normalised on import' : undefined + }, + // The U+2011 slug that never matched the website's filter. + ({ field, strapi, payload }) => + field === 'industry' && String(strapi).replace(/[‐-―]/g, '') === String(payload) + ? 'Non-breaking hyphen removed so the E-Commerce filter matches' + : undefined, +] + +function explain(diff: Omit): string | undefined { + for (const rule of EXPECTED) { + const reason = rule(diff) + if (reason) return reason + } + return undefined +} + +/** Fields that cannot match by construction: database identities and the + * timestamps a fresh import necessarily rewrites. */ +const IGNORED = new Set(['id', 'createdAt', 'updatedAt']) + +function compare(entity: string, key: string, strapi: Record, payload: Record, diffs: Diff[]) { + const fields = new Set([...Object.keys(strapi), ...Object.keys(payload)]) + for (const field of fields) { + if (IGNORED.has(field)) continue + const a = strapi[field] + const b = payload[field] + if (JSON.stringify(a) === JSON.stringify(b)) continue + + // Recurse one level so a nested image reports `image.src`, not `image`. + if (a && b && typeof a === 'object' && typeof b === 'object' && !Array.isArray(a) && !Array.isArray(b)) { + compare(entity, key, a as Record, b as Record, diffs) + continue + } + const diff = { entity, key, field, strapi: a, payload: b } + diffs.push({ ...diff, explained: explain(diff) }) + } +} + +async function collections(provider: CMSProvider) { + const [blogs, projects, gallery] = await Promise.all([ + provider.listBlogs({ limit: 1000 }), + provider.listProjects({ limit: 500 }), + provider.listGalleryItems({ limit: 1000 }), + ]) + return { + blogs: blogs.map((blog) => normalise(blog)), + projects: projects.map(normaliseProject), + gallery: gallery.map(normaliseGallery), + } +} + +async function main() { + const strapi = new StrapiProvider() + const payload = new PayloadProvider({ draft: false }) + + console.log('Reading Strapi…') + const fromStrapi = await collections(strapi) + console.log('Reading Payload…') + const fromPayload = await collections(payload) + + // fetchStrapi answers with an empty list when STRAPI_URL is unset, which would + // otherwise render as "every document is missing from Strapi" — a scary report + // about nothing. + const strapiTotal = fromStrapi.blogs.length + fromStrapi.projects.length + fromStrapi.gallery.length + if (strapiTotal === 0) { + console.error('Strapi returned no documents at all. Set STRAPI_URL and STRAPI_API_TOKEN — without them this comparison is meaningless.') + process.exit(2) + } + + const diffs: Diff[] = [] + const summary: Array> = [] + + for (const entity of ['blogs', 'projects', 'gallery'] as const) { + const left = fromStrapi[entity] as Array> + const right = fromPayload[entity] as Array> + const leftSlugs = left.map((item) => String(item.slug)) + const rightSlugs = right.map((item) => String(item.slug)) + + const missing = leftSlugs.filter((slug) => !rightSlugs.includes(slug)) + const extra = rightSlugs.filter((slug) => !leftSlugs.includes(slug)) + // Order is part of the rendered output: /projects and /gallery have no + // client-side sort, so the sequence the provider returns is what visitors see. + const orderMatches = JSON.stringify(leftSlugs) === JSON.stringify(rightSlugs) + + summary.push({ + entity, + strapi: left.length, + payload: right.length, + missingInPayload: missing.length, + onlyInPayload: extra.length, + orderMatches, + }) + + for (const slug of missing) diffs.push({ entity, key: slug, field: '(document)', strapi: 'present', payload: 'ABSENT' }) + for (const slug of extra) diffs.push({ entity, key: slug, field: '(document)', strapi: 'ABSENT', payload: 'present' }) + if (!orderMatches) { + diffs.push({ + entity, + key: '(ordering)', + field: 'sequence', + strapi: leftSlugs.slice(0, 8), + payload: rightSlugs.slice(0, 8), + }) + } + + const bySlug = new Map(right.map((item) => [String(item.slug), item])) + for (const item of left) { + const counterpart = bySlug.get(String(item.slug)) + if (counterpart) compare(entity, String(item.slug), item, counterpart, diffs) + } + } + + const unexplained = diffs.filter((diff) => !diff.explained) + const explained = diffs.filter((diff) => diff.explained) + + console.log('\nCounts and ordering') + console.table(summary) + + if (explained.length > 0) { + const reasons = explained.reduce>((counts, diff) => { + counts[diff.explained!] = (counts[diff.explained!] ?? 0) + 1 + return counts + }, {}) + console.log('\nIntended differences') + console.table(reasons) + } + + await mkdir('artifacts', { recursive: true }) + await writeFile(REPORT, JSON.stringify({ summary, unexplained, explained }, null, 2)) + + if (unexplained.length > 0) { + console.error(`\n${unexplained.length} unexplained difference(s) — full detail in ${REPORT}`) + for (const diff of unexplained.slice(0, 30)) { + console.error(` ${diff.entity}/${diff.key} ${diff.field}: strapi=${JSON.stringify(diff.strapi)?.slice(0, 90)} payload=${JSON.stringify(diff.payload)?.slice(0, 90)}`) + } + process.exitCode = 1 + } else { + console.log(`\nNo unexplained differences. Report written to ${REPORT}`) + } +} + +void main() diff --git a/wrangler.jsonc b/wrangler.jsonc index 30f67b6..02ccd2a 100644 --- a/wrangler.jsonc +++ b/wrangler.jsonc @@ -22,6 +22,19 @@ "observability": { "enabled": true, }, + // CMS selection. These are plaintext, not secrets: the tokens they pair with + // arrive through `wrangler secret bulk` in CI. Named environments do NOT + // inherit `vars`, so each env below repeats the full set. + // + // Draft reads need CMS_MODE=draft AND DEPLOYMENT_ROLE=preview together, so a + // single mistyped variable cannot expose drafts on the public domain. + "vars": { + "CMS_PROVIDER": "strapi", + "CMS_MODE": "published", + "DEPLOYMENT_ROLE": "development", + "PAYLOAD_URL": "https://cms.epyc.in", + }, + // Producer side of the contact-form webhook job. The consumer is a separate // worker — see workers/contact-webhook/. Named environments do NOT inherit // `queues`, so each env repeats its own producer binding below. @@ -45,8 +58,16 @@ }, ], "env": { + // Staging doubles as the protected content-preview site: it reads the + // latest drafts, serves noindex, and publishes no sitemap. "staging": { "name": "epyc-website-staging", + "vars": { + "CMS_PROVIDER": "strapi", + "CMS_MODE": "draft", + "DEPLOYMENT_ROLE": "preview", + "PAYLOAD_URL": "https://cms.epyc.in", + }, "d1_databases": [ { "binding": "DB", @@ -65,6 +86,12 @@ }, "production": { "name": "epyc-website-production", + "vars": { + "CMS_PROVIDER": "strapi", + "CMS_MODE": "published", + "DEPLOYMENT_ROLE": "production", + "PAYLOAD_URL": "https://cms.epyc.in", + }, "d1_databases": [ { "binding": "DB", From ed0236bc4d0d4d65e8ae438c6cb03c809e17143d Mon Sep 17 00:00:00 2001 From: Keshav Sharma Date: Sun, 23 Aug 2026 16:22:31 +0530 Subject: [PATCH 07/12] fix(cms): point the parity script at its actual filename The file was renamed from .mts to .ts so Node loads the imported providers as CommonJS, but the package script kept the old name. Typecheck passed because tsc does not read package.json scripts, so the break only surfaced on running it. Co-Authored-By: Claude Opus 5 (1M context) --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index e705f8d..925609f 100644 --- a/package.json +++ b/package.json @@ -12,7 +12,7 @@ "deploy:staging": "NEXT_PUBLIC_DEPLOY_ENV=staging opennextjs-cloudflare build && wrangler deploy --env staging && wrangler deploy --env staging --config workers/contact-webhook/wrangler.jsonc", "deploy:production": "NEXT_PUBLIC_DEPLOY_ENV=production opennextjs-cloudflare build && wrangler deploy --env production && wrangler deploy --env production --config workers/contact-webhook/wrangler.jsonc", "cf-typegen": "wrangler types --env-interface CloudflareEnv ./cloudflare-env.d.ts", - "cms:parity": "tsx scripts/cms-parity.mts" + "cms:parity": "tsx scripts/cms-parity.ts" }, "dependencies": { "@hookform/resolvers": "^5.2.2", From d08cfefb1582aa3cbcd5f5d4f6b5b7d55e952a85 Mon Sep 17 00:00:00 2001 From: Keshav Sharma Date: Sun, 23 Aug 2026 16:30:04 +0530 Subject: [PATCH 08/12] fix(cms): rebuild taxonomy shape, blank alt, and ordering ties MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The parity check found that PayloadProvider passed industry and platform through as bare slugs, where Strapi returned `{ title, slug }` and the normalisers read `.slug`. Every project therefore read as industry "other" and platform "website", which silently breaks filtering on /projects — 86 of 90 projects affected. A blank alt is now treated as absent so the consumers' `??` fallback to the document title actually fires; an empty string satisfied it and suppressed the alt entirely. Blog and project sorts gain legacyStrapiId as a tiebreaker, matching how Strapi ordered documents sharing a date. Co-Authored-By: Claude Opus 5 (1M context) --- lib/cms/payload-provider.test.ts | 27 ++++++++++++++++++++ lib/cms/payload-provider.ts | 44 ++++++++++++++++++++++++++++---- 2 files changed, 66 insertions(+), 5 deletions(-) diff --git a/lib/cms/payload-provider.test.ts b/lib/cms/payload-provider.test.ts index 7c04492..517cb76 100644 --- a/lib/cms/payload-provider.test.ts +++ b/lib/cms/payload-provider.test.ts @@ -114,6 +114,33 @@ describe('PayloadProvider requests', () => { }) describe('PayloadProvider field shapes', () => { + it('rebuilds industry and platform into the shape the normalisers read', async () => { + stubFetch([{ id: 1, title: 'P', slug: 'p', type: [], industry: 'community-initiative', platform: 'app', publishedAt: '2026-01-01' }]) + const [project] = await provider().listProjects() + + // Payload returns the bare slug where Strapi returned a relation. Passing it + // through unchanged makes every project read as industry "other". + expect(project.industry).toEqual({ title: 'Community Initiative', slug: 'community-initiative' }) + expect(project.platform).toEqual({ title: 'App', slug: 'app' }) + }) + + it('treats a blank alt as absent so consumers fall back to the title', async () => { + stubFetch([{ id: 1, title: 'Post', slug: 'post', updatedAt: '2026-01-01', coverImage: { ...mediaDoc, alt: ' ', alternativeText: null } }]) + const [blog] = await provider().listBlogs() + expect(blog.coverImage?.alternativeText).toBeNull() + }) + + it('breaks ordering ties the way Strapi did', async () => { + const calls = stubFetch([]) + await provider().listBlogs() + expect(calls[0].url.searchParams.get('sort')).toBe('-publishedDate,legacyStrapiId') + + vi.unstubAllGlobals() + const projects = stubFetch([]) + await provider().listProjects() + expect(projects[0].url.searchParams.get('sort')).toBe('-featured,-publishedAt,legacyStrapiId') + }) + it('keeps project type as an array and defaults gallery designers', async () => { stubFetch([{ id: 1, title: 'P', slug: 'p', type: ['WEBFLOW', 'SEO'], featured: true, publishedAt: '2026-01-01' }]) const [project] = await provider().listProjects() diff --git a/lib/cms/payload-provider.ts b/lib/cms/payload-provider.ts index 62e83ec..49ad874 100644 --- a/lib/cms/payload-provider.ts +++ b/lib/cms/payload-provider.ts @@ -17,7 +17,14 @@ type PayloadBlog = Omit & _status?: 'draft' | 'published' publishedAt?: string | null } -type PayloadProject = Omit & { id: string | number; thumbnail?: PayloadRelation } +type PayloadProject = Omit & { + id: string | number + thumbnail?: PayloadRelation + // Strapi modelled these as relations and returned objects; Payload stores + // selects and returns the bare value. + industry?: string | null + platform?: string | null +} type PayloadGallery = Omit & { id: string | number; image?: PayloadRelation } function populated(value: PayloadRelation | undefined): value is T { @@ -29,6 +36,19 @@ function size(value?: PayloadSize): MediaFormat | undefined { return { url: value.url, width: value.width, height: value.height } } +/** Strapi returned `{ title, slug }` for industry and platform, and the + * normalisers read `.slug`. Payload returns the slug alone, so handing it + * through unchanged silently collapses every project to industry "other" and + * platform "website". */ +function taxonomy(value?: string | null): { title: string; slug: string } | null { + if (!value) return null + const title = value + .split('-') + .map((word) => word.charAt(0).toUpperCase() + word.slice(1)) + .join(' ') + return { title, slug: value } +} + function mapMedia(value?: PayloadRelation): Media | null { if (!populated(value) || !value.url) return null // Consumers (`lib/blogs/normalise.ts`, `lib/projects/normalise.ts`) ask for @@ -45,7 +65,7 @@ function mapMedia(value?: PayloadRelation): Media | null { ...value, id: String(value.id), url: value.url, - alternativeText: value.alternativeText ?? value.alt ?? null, + alternativeText: value.alternativeText?.trim() || value.alt?.trim() || null, formats, } } @@ -99,7 +119,9 @@ export class PayloadProvider implements CMSProvider { async listBlogs(options: ListOptions = {}): Promise { const docs = await this.list('blogs', { - sort: '-publishedDate', + // Several posts share a publishedDate; Strapi ordered those by ascending + // numeric id, preserved on import as legacyStrapiId. + sort: '-publishedDate,legacyStrapiId', limit: String(options.limit ?? 100), ...(options.excludeSlug ? { 'where[slug][not_equals]': options.excludeSlug } : {}), }) @@ -112,8 +134,20 @@ export class PayloadProvider implements CMSProvider { } async listProjects(options: ListOptions = {}): Promise { - const docs = await this.list('projects', { sort: '-featured,-publishedAt', limit: String(options.limit ?? 200) }) - return docs.map((item) => ({ ...item, id: String(item.id), type: Array.isArray(item.type) ? item.type : [], thumbnail: mapMedia(item.thumbnail) })) + const docs = await this.list('projects', { + // legacyStrapiId breaks ties the way Strapi did — by ascending numeric id + // — which matters because most projects share a publishedAt. + sort: '-featured,-publishedAt,legacyStrapiId', + limit: String(options.limit ?? 200), + }) + return docs.map((item) => ({ + ...item, + id: String(item.id), + type: Array.isArray(item.type) ? item.type : [], + thumbnail: mapMedia(item.thumbnail), + industry: taxonomy(item.industry), + platform: taxonomy(item.platform), + })) } async listGalleryItems(options: ListOptions = {}): Promise { From 08464e05bc99c91de41652325ec6c7d83ae1c105 Mon Sep 17 00:00:00 2001 From: Keshav Sharma Date: Sun, 23 Aug 2026 17:22:15 +0530 Subject: [PATCH 09/12] feat(cms): read the plugin-seo meta group The CMS moves its SEO fields into plugin-seo's `meta` group. The provider flattens them back to metaTitle and metaDescription so pages, metadata and the parity comparison are unaffected. Co-Authored-By: Claude Opus 5 (1M context) --- lib/cms/payload-provider.test.ts | 16 ++++++++++++++++ lib/cms/payload-provider.ts | 7 ++++++- 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/lib/cms/payload-provider.test.ts b/lib/cms/payload-provider.test.ts index 517cb76..b1b46cf 100644 --- a/lib/cms/payload-provider.test.ts +++ b/lib/cms/payload-provider.test.ts @@ -124,6 +124,22 @@ describe('PayloadProvider field shapes', () => { expect(project.platform).toEqual({ title: 'App', slug: 'app' }) }) + it('flattens the plugin-seo meta group the way the pages consume it', async () => { + stubFetch([ + { + id: 1, title: 'Post', slug: 'post', updatedAt: '2026-01-01', + meta: { title: 'SEO title', description: 'SEO description' }, + }, + { id: 2, title: 'No meta', slug: 'no-meta', updatedAt: '2026-01-01' }, + ]) + const [withMeta, without] = await provider().listBlogs() + expect(withMeta.metaTitle).toBe('SEO title') + expect(withMeta.metaDescription).toBe('SEO description') + // generateMetadata falls back to the post title, so absent must be null + // rather than undefined-shaped noise. + expect(without.metaTitle).toBeNull() + }) + it('treats a blank alt as absent so consumers fall back to the title', async () => { stubFetch([{ id: 1, title: 'Post', slug: 'post', updatedAt: '2026-01-01', coverImage: { ...mediaDoc, alt: ' ', alternativeText: null } }]) const [blog] = await provider().listBlogs() diff --git a/lib/cms/payload-provider.ts b/lib/cms/payload-provider.ts index 49ad874..d456bac 100644 --- a/lib/cms/payload-provider.ts +++ b/lib/cms/payload-provider.ts @@ -10,8 +10,11 @@ type PayloadMedia = Omit & { sizes?: Record } type PayloadAuthor = Omit & { id: string | number; authorImage?: PayloadRelation } -type PayloadBlog = Omit & { +type PayloadBlog = Omit & { id: string | number + // plugin-seo groups the SEO fields; Strapi had them flat, and the website + // still consumes them flat. + meta?: { title?: string | null; description?: string | null; image?: PayloadRelation } | null coverImage?: PayloadRelation author?: PayloadRelation _status?: 'draft' | 'published' @@ -75,6 +78,8 @@ function mapBlog(item: PayloadBlog): Blog { ...item, id: String(item.id), publishedAt: item.publishedAt ?? item.publishedDate ?? item.updatedAt, + metaTitle: item.meta?.title ?? null, + metaDescription: item.meta?.description ?? null, coverImage: mapMedia(item.coverImage), author: populated(item.author) ? { ...item.author, id: String(item.author.id), authorImage: mapMedia(item.author.authorImage) } From e35a503078047667dbfbc03ee78384efeba46cdd Mon Sep 17 00:00:00 2001 From: Keshav Sharma Date: Thu, 27 Aug 2026 14:03:53 +0530 Subject: [PATCH 10/12] docs(cms): handoff brief for the Payload cutover Current state, the remaining steps with their acceptance checks, and the traps that have already cost time: the migration propagation race, the shared media bucket, generateSlug regenerating migrated slugs, omitted PATCH keys failing to clear values, and the pnpm version constraints. Co-Authored-By: Claude Opus 5 (1M context) --- docs/payload-cutover-handoff.md | 248 ++++++++++++++++++++++++++++++++ 1 file changed, 248 insertions(+) create mode 100644 docs/payload-cutover-handoff.md diff --git a/docs/payload-cutover-handoff.md b/docs/payload-cutover-handoff.md new file mode 100644 index 0000000..eded249 --- /dev/null +++ b/docs/payload-cutover-handoff.md @@ -0,0 +1,248 @@ +# Payload cutover — handoff + +Continuation brief for the Strapi → Payload migration. The design rationale is +in `docs/payload-cms-migration-plan.md`; this file is the current state, the +remaining work, and the traps that have already cost time. + +Written 2026-08-27, immediately after content parity came back clean. + +--- + +## 1. Where things stand + +**Content migration is complete and verified.** `pnpm cms:parity` compares what +the website would render from each CMS, on normalised output rather than raw API +responses, and reports: + +``` +blogs 29 / 29 order matches +projects 90 / 90 order matches +gallery 83 / 83 order matches +0 unexplained differences +``` + +The 390 reported differences are all declared as intended: media URLs now point +at originals instead of Strapi derivatives (Cloudflare resizes on request — +verified against the live zone), 8 duplicate service-tag spellings folded, and +2 corrected `e-commerce` slugs. + +**Nothing is switched over.** `CMS_PROVIDER=strapi` in every environment. +`epyc.in` and `staging.epyc.in` run the `main`/`production` branches, which +predate the provider abstraction entirely. + +### Deployed pieces + +| Thing | State | +|---|---| +| Payload CMS | `https://epyc-payload-cms.epyc.workers.dev` — live, CI-deployed | +| CMS repo | `teamEPYC/epyc-payload-cms`, branch `main` | +| Website repo | `teamEPYC/epyc-website-nextjs`, branch `feat/payload-cms-migration` (unmerged) | +| D1 | `epyc-payload-production` (`99aacd1b-3952-4436-a88e-261f72c4874b`), 5 migrations applied | +| R2 | `epyc-website-production` — **shared with the live site's media** | +| Cloudflare account | EPYC Production, `cbf6ee8739c8b48c351774cd83f1f413` (pinned in `wrangler.jsonc`) | +| Media host | `media.epyc.in` | +| Admin | `/admin` — one administrator account exists | + +`cms.epyc.in` is **still Strapi** and must stay that way until cutover. + +### Content in Payload + +170 media, 4 authors, 29 published blogs (+2 draft-only), 90 projects, 83 +gallery items. Every document carries `legacyStrapiDocumentId` (upsert key) and, +where ordering depends on it, `legacyStrapiId`. + +--- + +## 2. Remaining work, in order + +### Step 1 — Merge the website branch to `main` (agent) + +Deploys the provider abstraction to **staging only** (`main` → staging; +`production` branch → production). Both environments keep +`CMS_PROVIDER=strapi`, so this is expected to be visually a no-op — which is +exactly what makes it worth doing separately: it proves the refactor is safe +before any CMS switch. + +```bash +gh pr create --repo teamEPYC/epyc-website-nextjs \ + --base main --head feat/payload-cms-migration \ + --title "CMS provider abstraction and Payload provider" +``` + +**Acceptance:** staging renders identically to before. Spot-check `/blog`, +`/projects`, `/gallery`, one blog post, one gallery item. + +**Expected side effect:** staging becomes the content-preview deployment +(`CMS_MODE=draft`, `DEPLOYMENT_ROLE=preview`), so it serves Strapi drafts, +`noindex, nofollow`, `Cache-Control: private, no-store`, and an empty sitemap. +Step 2 must not lag behind this. + +### Step 2 — Protect staging (human, Cloudflare dashboard) + +Put Cloudflare Access in front of `staging.epyc.in`. It now serves unpublished +drafts. This cannot be done from code. + +**Acceptance:** an unauthenticated request to `staging.epyc.in` is challenged. + +### Step 3 — Preview credentials (human) + +1. Payload admin → Users → create a user with role **`preview`**. That role can + read drafts and cannot write anything — enforced in `lib/access.ts`, not by + convention. +2. Enable its API key, copy it. +3. Website repo → Settings → Secrets and variables → Actions: + - `STAGING_PAYLOAD_PREVIEW_TOKEN` — the key from above + - `CMS_REVALIDATION_SECRET` — must equal the value already set on the CMS + Worker (`wrangler secret list` shows it exists; the value is in the + password manager) + +`PRODUCTION_PAYLOAD_READ_TOKEN` is optional: published content is readable +anonymously. + +### Step 4 — Switch staging to Payload (agent) + +In the website repo's `wrangler.jsonc`, `env.staging.vars.CMS_PROVIDER`: +`strapi` → `payload`. Also update `CMS_PROVIDER` in the staging workflow's +OpenNext build env — it is needed at build time as well as at runtime, because +`robots.ts` and the root layout's metadata are evaluated during the build. + +**Acceptance, all on staging:** +- `/blog`, `/projects`, `/gallery` render, with counts 29 / 90 / 83 +- a blog post, a gallery item, and `/gallery/epyc-merchandise-tshirt-concept-design-(green-variant)` (parenthesised slug) all return 200 +- `/projects` industry filters work — every industry, not just "other" +- `curl -H "Accept: text/markdown" .../blog` returns markdown +- editing a draft in Payload changes staging within a minute; production is unaffected +- `robots.txt` disallows everything, `sitemap.xml` is empty + +### Step 5 — Production cutover (agent + human) + +Do this while the parity result is fresh; it decays as editors keep working in +Strapi. + +1. Announce an editorial freeze. +2. Re-run the export/import/parity cycle to capture anything edited since + 2026-08-27 (see §4). Parity must be clean again. +3. Merge `main` → `production`, with `env.production.vars.CMS_PROVIDER=payload` + and the same change in the production workflow's build env. +4. Verify on `epyc.in`: the acceptance list from step 4, plus no draft content + anywhere, and sitemap entries matching the previous count. +5. Make Strapi read-only for editors. **Do not delete anything.** + +### Step 6 — Observation window + +Keep Strapi read-only and the provider switch available for two weeks. Watch +CMS fetch failures, 404s on CMS-backed routes, webhook failures, and media +errors. Only then remove Strapi code, secrets and deployment config. + +--- + +## 3. Traps — read before touching anything + +**Migrations do not deploy themselves, and the naive sequence loses a race.** +`/api/db/migrate` runs the migrations bundled into the Worker, so a migration +can only be applied *after* its code is live. And `wrangler secret put` deploys +a new Worker version, so a request sent on the next line reaches the *previous* +version and answers "Migrations are not enabled". This cost three separate +debugging sessions, each surfacing as opaque 500s on every write. Always: + +```bash +pnpm cms:migrate # sets a one-time secret, waits for propagation, applies, removes it +``` + +**`/api/health` reports schema drift.** It compares bundled migrations against +`payload_migrations` and returns 503 `schema-behind` when they disagree. The +importer refuses to start in that state, and CI fails the deploy — a red CI run +straight after a migration-bearing deploy is the guard working, not a new +problem. Fix it by running `pnpm cms:migrate`. + +**The R2 bucket is shared with the live site.** Payload writes into +`epyc-website-production`, the bucket `media.epyc.in` already serves. This is +deliberate: imported images keep their existing URLs and nothing was copied. +The consequence is that **deleting an upload in the Payload admin deletes a live +site image.** + +**`generateSlug` defaults to true** and will regenerate a slug from the title, +discarding the real one. The importer sends `generateSlug: false`. Without it, +`breathewellbeing.in` becomes `breathewellbeingin` and three gallery slugs lose +their parentheses — and gallery slugs are live URLs. + +**An omitted key in a PATCH means "leave unchanged", not "clear".** Optional +fields go through `orNull()` in the importer. The earlier `?? undefined` version +published a draft's `caseStudyPath` onto the live document, because a null in +Strapi could not clear a value Payload already had. + +**Bump `MAPPER_VERSION` when changing importer field mapping.** Checkpoint keys +include it, so documents are re-imported rather than skipped holding values from +the previous mapping. Media keys deliberately exclude it — uploads do not +change, and re-uploading 170 images is slow. + +**pnpm must be 11.x.** The lockfile records the Payload patch with a hash pnpm +10 rejects; `packageManager` in `package.json` pins it. `pnpm-workspace.yaml` +must also keep `allowBuilds` entries as real booleans — the scaffold's +placeholder strings fail CI installs. + +**The Payload CLI needs its wrapper scripts.** `payload.config.ts` exports a +promise from an async IIFE and imports Lexical lazily, and there is a patch for +Payload's `@next/env` import. Use `pnpm db:generate` / `pnpm generate:types`, +never `npx payload` directly. + +**CI never sets Worker secrets.** `PAYLOAD_SECRET`, `CMS_REVALIDATION_SECRET` +and friends are set once by hand and must not be added to GitHub Actions. + +**Do not point `cms.epyc.in` at Payload before cutover.** It is the live Strapi +and the production site reads from it. + +--- + +## 4. Commands + +```bash +# CMS repo +pnpm cms:migrate # apply pending migrations, safely +PAYLOAD_API_KEY=... pnpm cms:import-payload # idempotent; --dry-run to plan only +STRAPI_URL=... STRAPI_API_TOKEN=... pnpm cms:export-strapi +pnpm db:generate # new migration after a schema change + +# Website repo +STRAPI_URL=... STRAPI_API_TOKEN=... PAYLOAD_URL=... pnpm cms:parity + +# State checks +curl -s https://epyc-payload-cms.epyc.workers.dev/api/health +pnpm exec wrangler d1 execute epyc-payload-production --remote \ + --command "SELECT (SELECT COUNT(*) FROM blogs) blogs, (SELECT COUNT(*) FROM projects) projects, (SELECT COUNT(*) FROM gallery) gallery" +``` + +The import needs an API key belonging to an `administrator` or `editor`; the +`preview` role cannot write. Disable that key again once the import is done. + +--- + +## 5. Rollback + +The switch is a variable, not a revert: + +1. Set `CMS_PROVIDER=strapi` in the affected environment's `wrangler.jsonc` + **and** in that workflow's build env, then deploy. +2. Purge or revalidate the CMS-backed routes. +3. Re-enable Strapi editing if it was made read-only. +4. Leave Payload and its data intact for diagnosis. + +Payload's D1 can be restored via Time Travel (`wrangler d1 time-travel info` / +`restore`) if a schema or data failure requires it. Strapi's database and media +must not be deleted or modified destructively during the observation window. + +--- + +## 6. After cutover + +- **Lexical rich text.** `content` is a `code` field holding CKEditor HTML so the + migration could be verified byte for byte. The conversion plan, including the + four pieces of markup needing review, is in the CMS repo's + `docs/OPERATIONS.md`. Convert `content` to Lexical and generate HTML in a save + hook, so the website and the parity checker keep reading HTML. +- **Failure visibility.** A CMS outage currently renders an empty page as a + successful response. `PayloadProvider` already throws; the Strapi client's + empty-list fallback should be tightened or removed once Strapi is gone. +- **Author drafts.** The Authors collection keeps no version history, so the + unpublished draft edits on `keshav-sharma` were not migrated. Enable versions + there if that matters. From d8dce4636e458747eaa5e85d10b2870ab583bec8 Mon Sep 17 00:00:00 2001 From: Keshav Sharma <80240712+keshav-epyc@users.noreply.github.com> Date: Thu, 27 Aug 2026 14:55:54 +0530 Subject: [PATCH 11/12] chore(cms): switch staging to Payload --- .github/workflows/deploy-staging.yml | 4 ++-- wrangler.jsonc | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 75a0d88..448a61c 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -48,10 +48,10 @@ jobs: # Staging is the content-preview site: draft reads, noindex, no sitemap. # robots.ts and the root layout's metadata are evaluated at build time, # so these must be present here as well as in the Worker's vars. - CMS_PROVIDER: strapi + CMS_PROVIDER: payload CMS_MODE: draft DEPLOYMENT_ROLE: preview - PAYLOAD_URL: https://cms.epyc.in + PAYLOAD_URL: https://epyc-payload-cms.epyc.workers.dev PAYLOAD_PREVIEW_TOKEN: ${{ secrets.STAGING_PAYLOAD_PREVIEW_TOKEN }} CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} diff --git a/wrangler.jsonc b/wrangler.jsonc index 02ccd2a..bf4128b 100644 --- a/wrangler.jsonc +++ b/wrangler.jsonc @@ -63,10 +63,10 @@ "staging": { "name": "epyc-website-staging", "vars": { - "CMS_PROVIDER": "strapi", + "CMS_PROVIDER": "payload", "CMS_MODE": "draft", "DEPLOYMENT_ROLE": "preview", - "PAYLOAD_URL": "https://cms.epyc.in", + "PAYLOAD_URL": "https://epyc-payload-cms.epyc.workers.dev", }, "d1_databases": [ { @@ -109,4 +109,4 @@ }, }, }, -} \ No newline at end of file +} From 5f10fc8a05d3e2504ad83c3f5ad17e831269aab8 Mon Sep 17 00:00:00 2001 From: Keshav Sharma <80240712+keshav-epyc@users.noreply.github.com> Date: Thu, 27 Aug 2026 15:07:26 +0530 Subject: [PATCH 12/12] chore(cms): switch production to Payload --- .github/workflows/deploy-production.yml | 4 ++-- wrangler.jsonc | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/deploy-production.yml b/.github/workflows/deploy-production.yml index 4986d7b..b7c23b5 100644 --- a/.github/workflows/deploy-production.yml +++ b/.github/workflows/deploy-production.yml @@ -48,10 +48,10 @@ jobs: # Published-only reads. robots.ts and the root layout's metadata are # evaluated at build time, so these must be present here as well as in # the Worker's vars. - CMS_PROVIDER: strapi + CMS_PROVIDER: payload CMS_MODE: published DEPLOYMENT_ROLE: production - PAYLOAD_URL: https://cms.epyc.in + PAYLOAD_URL: https://epyc-payload-cms.epyc.workers.dev PAYLOAD_READ_TOKEN: ${{ secrets.PRODUCTION_PAYLOAD_READ_TOKEN }} CMS_REVALIDATION_SECRET: ${{ secrets.CMS_REVALIDATION_SECRET }} diff --git a/wrangler.jsonc b/wrangler.jsonc index bf4128b..518ea46 100644 --- a/wrangler.jsonc +++ b/wrangler.jsonc @@ -87,10 +87,10 @@ "production": { "name": "epyc-website-production", "vars": { - "CMS_PROVIDER": "strapi", + "CMS_PROVIDER": "payload", "CMS_MODE": "published", "DEPLOYMENT_ROLE": "production", - "PAYLOAD_URL": "https://cms.epyc.in", + "PAYLOAD_URL": "https://epyc-payload-cms.epyc.workers.dev", }, "d1_databases": [ {