056c47581f
Ships the second dashboard surface — a Pattern Library + Preview Theatre — that presents the 4-section PDP pilot batch back to Umar, compliance, and the board in an editorial format. Adds the full data layer that drives it: 5 source-backed per-SKU drafts at QA 100/100, 15 competitor PDP semantic extracts, PubMed evidence packs, EFSA claims library extension, JV brand voice guide, hand-curated product FAQs, and the Matrixify-ready CSV exports for Lewis. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
93 lines
3.5 KiB
TypeScript
93 lines
3.5 KiB
TypeScript
#!/usr/bin/env bun
|
|
import { mkdirSync, writeFileSync } from 'fs'
|
|
import { dirname, join } from 'path'
|
|
|
|
const root = process.cwd()
|
|
const rawRoot = join(root, 'data', 'sources', 'apify', 'raw')
|
|
const token = Bun.env.APIFY_TOKEN || Bun.env.APIFY_API_TOKEN
|
|
const generatedAt = new Date().toISOString()
|
|
if (!token) throw new Error('Missing APIFY_TOKEN or APIFY_API_TOKEN')
|
|
|
|
const maxTotalChargeUsd = Number(Bun.env.APIFY_HEALTHSPAN_CONTENT_MAX_USD || 0.05)
|
|
|
|
function writeJson(path: string, data: any) {
|
|
mkdirSync(dirname(path), { recursive: true })
|
|
writeFileSync(path, JSON.stringify(data, null, 2) + '\n', 'utf8')
|
|
}
|
|
|
|
async function apify(path: string, options: RequestInit = {}) {
|
|
const sep = path.includes('?') ? '&' : '?'
|
|
const res = await fetch(`https://api.apify.com${path}${sep}token=${encodeURIComponent(token!)}`, {
|
|
...options,
|
|
headers: { 'Content-Type': 'application/json', ...(options.headers || {}) }
|
|
})
|
|
const text = await res.text()
|
|
let data: any
|
|
try { data = text ? JSON.parse(text) : null } catch { data = text }
|
|
if (!res.ok) throw new Error(`Apify ${res.status} ${res.statusText}: ${typeof data === 'string' ? data : JSON.stringify(data)}`)
|
|
return data
|
|
}
|
|
|
|
async function waitForRun(runId: string, timeoutMs = 1000 * 60 * 8) {
|
|
const started = Date.now()
|
|
let run = (await apify(`/v2/actor-runs/${runId}`)).data
|
|
while (!['SUCCEEDED', 'FAILED', 'ABORTED', 'TIMED-OUT'].includes(run.status)) {
|
|
if (Date.now() - started > timeoutMs) throw new Error(`Timed out waiting for ${runId}`)
|
|
await new Promise(resolve => setTimeout(resolve, 5000))
|
|
run = (await apify(`/v2/actor-runs/${runId}`)).data
|
|
console.log(`${runId}: ${run.status} usd=${run.usageTotalUsd || 0}`)
|
|
}
|
|
return run
|
|
}
|
|
|
|
const input = {
|
|
startUrls: [{ url: 'https://www.healthspan.co.uk/optivision/' }],
|
|
maxCrawlPages: 1,
|
|
maxCrawlDepth: 0,
|
|
crawlerType: 'cheerio',
|
|
removeElementsCssSelector: 'script, style, noscript, svg',
|
|
saveHtml: false,
|
|
saveMarkdown: true,
|
|
saveScreenshots: false,
|
|
proxyConfiguration: { useApifyProxy: true }
|
|
}
|
|
|
|
console.log(`Starting apify/website-content-crawler for Healthspan OptiVision, maxCrawlPages=1, cap=$${maxTotalChargeUsd}`)
|
|
const start = await apify(`/v2/acts/apify~website-content-crawler/runs?maxTotalChargeUsd=${maxTotalChargeUsd}`, {
|
|
method: 'POST',
|
|
body: JSON.stringify(input)
|
|
})
|
|
const run = await waitForRun(start.data.id)
|
|
const items = run.defaultDatasetId ? await apify(`/v2/datasets/${run.defaultDatasetId}/items?clean=true`) : []
|
|
const result = {
|
|
capturedAt: generatedAt,
|
|
job: 'healthspan-content-crawler',
|
|
actor: 'apify/website-content-crawler',
|
|
maxTotalChargeUsd,
|
|
input,
|
|
run: {
|
|
id: run.id,
|
|
status: run.status,
|
|
statusMessage: run.statusMessage,
|
|
defaultDatasetId: run.defaultDatasetId,
|
|
startedAt: run.startedAt,
|
|
finishedAt: run.finishedAt,
|
|
usageTotalUsd: run.usageTotalUsd,
|
|
usage: run.usage,
|
|
stats: run.stats
|
|
},
|
|
itemCount: Array.isArray(items) ? items.length : 0,
|
|
items: Array.isArray(items) ? items.map((item: any) => ({
|
|
...item,
|
|
sku: 'JV-VISISOFT',
|
|
sourceType: 'competitor_pdp_content_crawler',
|
|
label: 'Healthspan OptiVision content crawler',
|
|
competitor: 'Healthspan'
|
|
})) : items
|
|
}
|
|
|
|
const stamp = generatedAt.replace(/[:.]/g, '-')
|
|
writeJson(join(rawRoot, `top3-specialist-healthspan-content-crawler-${stamp}.json`), result)
|
|
writeJson(join(rawRoot, 'top3-specialist-healthspan-content-crawler-latest.json'), result)
|
|
console.log(`Healthspan content crawler ${run.status}: ${result.itemCount} items, usd=${run.usageTotalUsd || 0}`)
|