From 1476ecea4c84da61c546874efc75b0f59dc6ae12 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 15 Sep 2026 12:05:20 +0000 Subject: [PATCH 1/3] Remove 79 compiled files I committed by accident MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit 418d7a7 added apps/worker/test-dist-q/ — 79 transpiled .js files, 624 KB — and it merged to main via #41. I used `git add -A`, and this .gitignore read exactly `test-dist/`, so a scratch TypeScript out-dir named `test-dist-q` went through a gap one character wide. Nothing generates them: no tsconfig, npm script, shell script or workflow mentions `test-dist-q`, and the only declared out-dirs are `test-dist` (tsconfig.test.json) and `dist` (packages/worker-runner). Nothing imports them either. There is no production impact — wrangler.toml bundles from `main = "src/index.ts"`, the test suite compiles to test-dist/, eslint runs as `eslint 'src/**/*.ts'`, and the CI gates scan ':(glob)apps/worker/src/**/*.ts'. The cost was scanner noise. There is no CodeQL workflow in this repo: the Analyze checks are GitHub default setup, configured in repo settings, and they scan the whole tree with no path exclusions. So a code-quality bot reviewed generated code on #41 and reported a "useless conditional" at test-dist-q/entities/profile.js:49 — the transpiled retry guard `attempt < 5 && !acquired`, which is fine in the source it came from. That would have recurred on every future PR, and the files would have gone on drifting from src/ and polluting every grep of the repo. The ignore rule is now a glob rather than a literal, so the next scratch out-dir cannot repeat this. Verified it still covers plain `test-dist/`, and that `test-dist-q` was the only committed build output in the tree — packages/worker-runner/dist is untracked. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/.gitignore | 6 +- apps/worker/test-dist-q/ai/budget.js | 46 -- apps/worker/test-dist-q/ai/cache.js | 33 - apps/worker/test-dist-q/ai/extract.js | 335 --------- apps/worker/test-dist-q/analytics/events.js | 79 -- apps/worker/test-dist-q/db/error_log.js | 125 ---- apps/worker/test-dist-q/entities/channels.js | 61 -- apps/worker/test-dist-q/entities/dualwrite.js | 375 ---------- apps/worker/test-dist-q/entities/facts.js | 282 -------- apps/worker/test-dist-q/entities/garbage.js | 652 ----------------- apps/worker/test-dist-q/entities/model.js | 6 - apps/worker/test-dist-q/entities/normalize.js | 120 --- .../entities/profile-predicates.js | 218 ------ .../test-dist-q/entities/profile-shapes.js | 4 - apps/worker/test-dist-q/entities/profile.js | 681 ------------------ apps/worker/test-dist-q/entities/query.js | 154 ---- apps/worker/test-dist-q/entities/roles.js | 120 --- apps/worker/test-dist-q/entities/summary.js | 121 ---- .../test-dist-q/entities/summaryQueue.js | 26 - apps/worker/test-dist-q/entities/tags.js | 37 - apps/worker/test-dist-q/errors.js | 328 --------- apps/worker/test-dist-q/personas/repo.js | 403 ----------- apps/worker/test-dist-q/personas/score.js | 281 -------- .../test-dist-q/scraper/firms_upsert.js | 267 ------- apps/worker/test-dist-q/scraper/normalize.js | 104 --- .../scraper/parsers/firmlists/types.js | 1 - apps/worker/test-dist-q/scraper/rateLimit.js | 53 -- .../services/personaMatchTrigger.js | 59 -- .../test-dist-q/services/personaMatching.js | 484 ------------- .../services/personaMatchingScorers.js | 366 ---------- .../services/personas/kinds/_generic.js | 45 -- .../personas/kinds/academic_researcher.js | 7 - .../personas/kinds/account_company.js | 17 - .../services/personas/kinds/acquirer.js | 7 - .../personas/kinds/angel_individual.js | 7 - .../services/personas/kinds/beta_tester.js | 7 - .../services/personas/kinds/buyer_person.js | 37 - .../personas/kinds/channel_partner.js | 7 - .../personas/kinds/co_founder_match.js | 7 - .../services/personas/kinds/competitor.js | 7 - .../services/personas/kinds/design_partner.js | 7 - .../personas/kinds/engineering_hire.js | 7 - .../services/personas/kinds/executive_hire.js | 7 - .../services/personas/kinds/founder.js | 68 -- .../personas/kinds/fractional_executive.js | 7 - .../kinds/government_grant_officer.js | 7 - .../services/personas/kinds/index.js | 71 -- .../personas/kinds/integration_partner.js | 7 - .../services/personas/kinds/investor_firm.js | 64 -- .../personas/kinds/investor_person.js | 14 - .../personas/kinds/journalist_analyst.js | 7 - .../personas/kinds/limited_partner.js | 7 - .../services/personas/kinds/policy_advisor.js | 7 - .../services/personas/kinds/regulator.js | 7 - .../personas/kinds/service_provider.js | 7 - .../services/personas/kinds/taxonomy.js | 86 --- .../services/personas/kinds/thought_leader.js | 7 - .../personas/kinds/venture_partner.js | 50 -- .../services/relationships/_safeQuery.js | 18 - .../services/relationships/baselines.js | 42 -- .../extractors/advisorFromBio.js | 38 - .../extractors/boardSeatFromFilings.js | 78 -- .../extractors/coAuthorFromPublications.js | 35 - .../extractors/coInvestorFromDeals.js | 42 -- .../extractors/colleagueOverlap.js | 45 -- .../extractors/educationFromBio.js | 36 - .../employmentHistoryFromLinkedIn.js | 44 -- .../extractors/familyFromPublicSources.js | 40 - .../extractors/investedInFromDeals.js | 49 -- .../extractors/mentionFromNews.js | 40 - .../extractors/portfolioFromFirmSite.js | 47 -- .../relationships/extractors/schoolWith.js | 35 - .../extractors/worksAtFromTitle.js | 46 -- .../services/relationships/orchestrator.js | 116 --- .../services/relationships/persist.js | 92 --- .../services/relationships/resolve.js | 58 -- .../services/relationships/types.js | 2 - .../test-dist-q/services/roleInference.js | 123 ---- .../test-dist-q/services/secEdgar/xref.js | 128 ---- apps/worker/test-dist-q/types.js | 1 - 80 files changed, 5 insertions(+), 7562 deletions(-) delete mode 100644 apps/worker/test-dist-q/ai/budget.js delete mode 100644 apps/worker/test-dist-q/ai/cache.js delete mode 100644 apps/worker/test-dist-q/ai/extract.js delete mode 100644 apps/worker/test-dist-q/analytics/events.js delete mode 100644 apps/worker/test-dist-q/db/error_log.js delete mode 100644 apps/worker/test-dist-q/entities/channels.js delete mode 100644 apps/worker/test-dist-q/entities/dualwrite.js delete mode 100644 apps/worker/test-dist-q/entities/facts.js delete mode 100644 apps/worker/test-dist-q/entities/garbage.js delete mode 100644 apps/worker/test-dist-q/entities/model.js delete mode 100644 apps/worker/test-dist-q/entities/normalize.js delete mode 100644 apps/worker/test-dist-q/entities/profile-predicates.js delete mode 100644 apps/worker/test-dist-q/entities/profile-shapes.js delete mode 100644 apps/worker/test-dist-q/entities/profile.js delete mode 100644 apps/worker/test-dist-q/entities/query.js delete mode 100644 apps/worker/test-dist-q/entities/roles.js delete mode 100644 apps/worker/test-dist-q/entities/summary.js delete mode 100644 apps/worker/test-dist-q/entities/summaryQueue.js delete mode 100644 apps/worker/test-dist-q/entities/tags.js delete mode 100644 apps/worker/test-dist-q/errors.js delete mode 100644 apps/worker/test-dist-q/personas/repo.js delete mode 100644 apps/worker/test-dist-q/personas/score.js delete mode 100644 apps/worker/test-dist-q/scraper/firms_upsert.js delete mode 100644 apps/worker/test-dist-q/scraper/normalize.js delete mode 100644 apps/worker/test-dist-q/scraper/parsers/firmlists/types.js delete mode 100644 apps/worker/test-dist-q/scraper/rateLimit.js delete mode 100644 apps/worker/test-dist-q/services/personaMatchTrigger.js delete mode 100644 apps/worker/test-dist-q/services/personaMatching.js delete mode 100644 apps/worker/test-dist-q/services/personaMatchingScorers.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/_generic.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/account_company.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/acquirer.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/angel_individual.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/beta_tester.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/buyer_person.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/channel_partner.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/competitor.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/design_partner.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/executive_hire.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/founder.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/index.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/integration_partner.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/investor_firm.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/investor_person.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/limited_partner.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/regulator.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/service_provider.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/taxonomy.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/thought_leader.js delete mode 100644 apps/worker/test-dist-q/services/personas/kinds/venture_partner.js delete mode 100644 apps/worker/test-dist-q/services/relationships/_safeQuery.js delete mode 100644 apps/worker/test-dist-q/services/relationships/baselines.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/advisorFromBio.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/boardSeatFromFilings.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/coAuthorFromPublications.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/coInvestorFromDeals.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/colleagueOverlap.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/educationFromBio.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/employmentHistoryFromLinkedIn.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/familyFromPublicSources.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/investedInFromDeals.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/mentionFromNews.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/portfolioFromFirmSite.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/schoolWith.js delete mode 100644 apps/worker/test-dist-q/services/relationships/extractors/worksAtFromTitle.js delete mode 100644 apps/worker/test-dist-q/services/relationships/orchestrator.js delete mode 100644 apps/worker/test-dist-q/services/relationships/persist.js delete mode 100644 apps/worker/test-dist-q/services/relationships/resolve.js delete mode 100644 apps/worker/test-dist-q/services/relationships/types.js delete mode 100644 apps/worker/test-dist-q/services/roleInference.js delete mode 100644 apps/worker/test-dist-q/services/secEdgar/xref.js delete mode 100644 apps/worker/test-dist-q/types.js diff --git a/apps/worker/.gitignore b/apps/worker/.gitignore index d5677e91..72d9a501 100644 --- a/apps/worker/.gitignore +++ b/apps/worker/.gitignore @@ -1 +1,5 @@ -test-dist/ +# Compiled test output. The glob is deliberate: this rule was exactly +# `test-dist/`, and a scratch out-dir named `test-dist-q` was committed +# through the gap by a `git add -A` — 79 generated files that GitHub's +# default-setup CodeQL then scanned and filed findings against. +test-dist*/ diff --git a/apps/worker/test-dist-q/ai/budget.js b/apps/worker/test-dist-q/ai/budget.js deleted file mode 100644 index 3cfcadb5..00000000 --- a/apps/worker/test-dist-q/ai/budget.js +++ /dev/null @@ -1,46 +0,0 @@ -// AI / Vectorize daily budget caps (Task #25 step 10). -// -// Reads AI_DAILY_NEURONS_CAP and VECTORIZE_DAILY_QUERIES_CAP from env vars. -// Polls the D1 ai_cost_daily roll-up (cached 60s in KV) for the running -// total; refuses calls past the cap so a runaway loop can't drain the -// account. /api/scrapers/health surfaces burn-down for the dashboard. -const KV_KEY = "ai-budget:today"; -const CACHE_TTL = 60; -export async function getBurn(env) { - const cached = await env.SCRAPE_CACHE?.get(KV_KEY); - if (cached) { - try { - return JSON.parse(cached); - } - catch { /* fall through */ } - } - const day = new Date().toISOString().slice(0, 10); - const totals = await env.DB.prepare(`SELECT - SUM(neurons) AS neurons, - SUM(cost_usd) AS cost, - SUM(CASE WHEN purpose LIKE 'vectorize_%' THEN calls ELSE 0 END) AS vec_calls - FROM ai_cost_daily WHERE day = ?`).bind(day).first().catch(() => null); - const snap = { - day, - neurons_used: Number(totals?.neurons ?? 0), - neurons_cap: Number(env.AI_DAILY_NEURONS_CAP ?? "0") || 0, - vectorize_used: Number(totals?.vec_calls ?? 0), - vectorize_cap: Number(env.VECTORIZE_DAILY_QUERIES_CAP ?? "0") || 0, - cost_usd: Number(totals?.cost ?? 0), - }; - try { - await env.SCRAPE_CACHE?.put(KV_KEY, JSON.stringify(snap), { expirationTtl: CACHE_TTL }); - } - catch { /* best-effort */ } - return snap; -} -export async function assertBudget(env, kind) { - const snap = await getBurn(env); - if (kind === "ai" && snap.neurons_cap > 0 && snap.neurons_used >= snap.neurons_cap) { - return { ok: false, reason: `neurons_cap_reached:${snap.neurons_used}/${snap.neurons_cap}` }; - } - if (kind === "vectorize" && snap.vectorize_cap > 0 && snap.vectorize_used >= snap.vectorize_cap) { - return { ok: false, reason: `vectorize_cap_reached:${snap.vectorize_used}/${snap.vectorize_cap}` }; - } - return { ok: true }; -} diff --git a/apps/worker/test-dist-q/ai/cache.js b/apps/worker/test-dist-q/ai/cache.js deleted file mode 100644 index 850329c6..00000000 --- a/apps/worker/test-dist-q/ai/cache.js +++ /dev/null @@ -1,33 +0,0 @@ -const PREFIX = "ai-cache"; -export async function sha256Hex(input) { - const buf = new TextEncoder().encode(input); - const digest = await crypto.subtle.digest("SHA-256", buf); - return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); -} -export async function aiCacheGet(env, key) { - if (!env.AI_CACHE) - return null; - try { - const obj = await env.AI_CACHE.get(`${PREFIX}/${key}`); - if (!obj) - return null; - const text = await obj.text(); - return JSON.parse(text); - } - catch { - return null; - } -} -export async function aiCachePut(env, key, value) { - if (!env.AI_CACHE) - return; - try { - await env.AI_CACHE.put(`${PREFIX}/${key}`, JSON.stringify(value), { - httpMetadata: { contentType: "application/json" }, - customMetadata: { stored_at: new Date().toISOString() }, - }); - } - catch { - /* swallow — cache is best-effort */ - } -} diff --git a/apps/worker/test-dist-q/ai/extract.js b/apps/worker/test-dist-q/ai/extract.js deleted file mode 100644 index fdde5b67..00000000 --- a/apps/worker/test-dist-q/ai/extract.js +++ /dev/null @@ -1,335 +0,0 @@ -// AI-powered extraction (Task #25 step 2). -// -// Strategy: deterministic strategies in `firmcrawl/personExtract.ts` run -// first (cheap). Misses get routed through Workers AI with a strict JSON -// schema response, chunked at ~6KB. Every call is cached by -// sha256(model+prompt+chunk) in the AI_CACHE R2 bucket (30-day TTL via -// bucket lifecycle policy). Verification pass drops <0.6 confidence. -// -// Hooked into pipeline.ts firm_team_crawl path opportunistically: if `AI` -// binding is present and deterministic strategies returned <3 people on a -// non-trivial page, we run the AI pass and union by nameKey. -import { aiCacheGet, aiCachePut, sha256Hex } from "./cache"; -import { assertBudget } from "./budget"; -import { limitAi } from "../scraper/rateLimit"; -import { trackAi } from "../analytics/events"; -// Task #2: hard timeout for Workers AI calls. The binding does not accept -// AbortSignal, so we race against a timer and surface a uniform -// "ai_timeout" error. -// -// POLICY: one canonical 30s ceiling for any single AI call (constant -// `AI_TIMEOUT_MS` below). This is well below the 90s default job -// budget, so even three serial AI calls fit inside a single job's -// wall-clock ceiling. Short-form purposes (embeddings, arbitration) -// use `AI_TIMEOUT_SHORT_MS` (20s) since they're trivially smaller. -// -// NB: this is a *caller-side* timeout — it bounds how long the worker -// will wait on the Workers AI binding, but it does NOT cancel the -// underlying model execution. The binding doesn't expose an -// AbortSignal as of this revision, so model inference may continue -// (and bill) for a short tail after we move on. Acceptable today -// because the queue-level budget + sweeper will reclaim the job; if -// the binding gains cancellation, swap the race for a real abort. -const AI_TIMEOUT_MS = 30_000; -const AI_TIMEOUT_SHORT_MS = 20_000; -async function runAiWithTimeout(p, ms, label) { - let timer = null; - try { - return await Promise.race([ - p, - new Promise((_, reject) => { - timer = setTimeout(() => reject(new Error(`ai_timeout:${label}:${ms}ms`)), ms); - }), - ]); - } - finally { - if (timer) - clearTimeout(timer); - } -} -const PERSON_SCHEMA = { - type: "object", - properties: { - people: { - type: "array", - items: { - type: "object", - properties: { - name: { type: "string" }, - role: { type: "string" }, - email: { type: "string" }, - linkedin: { type: "string" }, - twitter: { type: "string" }, - bio: { type: "string" }, - confidence: { type: "number" }, - }, - required: ["name", "confidence"], - }, - }, - }, - required: ["people"], -}; -const CHUNK_BYTES = 6000; -const MIN_CONFIDENCE = 0.6; -function chunk(text, size) { - const out = []; - for (let i = 0; i < text.length; i += size) - out.push(text.slice(i, i + size)); - return out; -} -function stripHtml(html) { - return html - .replace(/]*>[\s\S]*?<\/script>/gi, " ") - .replace(/]*>[\s\S]*?<\/style>/gi, " ") - .replace(/<[^>]+>/g, " ") - .replace(/\s+/g, " ") - .trim(); -} -export async function aiExtractPeople(env, html, jobId) { - if (!env.AI) - return []; - const ok = await assertBudget(env, "ai"); - if (!ok.ok) - return []; - if (!(await limitAi(env))) - return []; - const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; - const text = stripHtml(html); - const chunks = chunk(text, CHUNK_BYTES).slice(0, 4); // hard ceiling per page - const all = []; - for (const c of chunks) { - const cacheKey = await sha256Hex(`${model}:people:${c}`); - const cached = await aiCacheGet(env, cacheKey); - if (cached) { - trackAi(env, { purpose: "extraction", model, cacheHit: true, jobId }); - all.push(...cached); - continue; - } - const t0 = Date.now(); - let people = []; - try { - const res = (await runAiWithTimeout(env.AI.run(model, { - messages: [ - { role: "system", content: "Extract investors/partners as JSON. Skip non-people. Return strict JSON." }, - { role: "user", content: `Extract people from this team-page text. ${c}` }, - ], - response_format: { type: "json_schema", json_schema: PERSON_SCHEMA }, - }), AI_TIMEOUT_MS, "extract_people")); - const parsed = parsePeopleResponse(res); - people = parsed.filter((p) => (p.confidence ?? 0) >= MIN_CONFIDENCE); - } - catch (e) { - console.warn("aiExtractPeople failed", e.message); - } - trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(c.length), jobId }); - await aiCachePut(env, cacheKey, people); - all.push(...people); - } - return dedupePeopleByName(all); -} -function parsePeopleResponse(res) { - const r = res; - if (Array.isArray(r?.people)) - return r.people.filter((p) => p && typeof p.name === "string"); - if (typeof r?.response === "string") { - try { - const j = JSON.parse(r.response); - if (Array.isArray(j?.people)) - return j.people.filter((p) => p && typeof p.name === "string"); - } - catch { /* fall through */ } - } - return []; -} -function dedupePeopleByName(arr) { - const map = new Map(); - for (const p of arr) { - const key = p.name.trim().toLowerCase(); - if (!key) - continue; - const cur = map.get(key); - if (!cur || (p.confidence ?? 0) > (cur.confidence ?? 0)) - map.set(key, p); - } - return [...map.values()]; -} -// Rough neurons estimate: tokens ≈ chars/4, llama-3.1-8b ~ 0.011 neurons/token. -function estimateNeurons(chars) { - const tokens = Math.ceil(chars / 4); - return Math.round(tokens * 0.011 * 1000) / 1000; -} -const TABLE_SCHEMA = { - type: "object", - properties: { - tables: { - type: "array", - items: { - type: "object", - properties: { - headers: { type: "array", items: { type: "string" } }, - rows: { type: "array", items: { type: "array", items: { type: "string" } } }, - }, - required: ["headers", "rows"], - }, - }, - }, - required: ["tables"], -}; -const TABLE_PAGE_CHAR_CAP = 8000; -const MAX_AI_PAGES = 12; -export async function aiExtractTablesFromPdfPages(env, pageTexts) { - if (!env.AI || !pageTexts.length) - return []; - const ok = await assertBudget(env, "ai"); - if (!ok.ok) - return []; - const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; - const out = []; - let lastHeaderKey = null; - const pages = pageTexts.slice(0, MAX_AI_PAGES); - for (let p = 0; p < pages.length; p++) { - const raw = pages[p].trim(); - if (raw.length < 40) - continue; - const text = raw.length > TABLE_PAGE_CHAR_CAP ? raw.slice(0, TABLE_PAGE_CHAR_CAP) : raw; - const cacheKey = await sha256Hex(`${model}:pdf-tables:${text}`); - let pageTables = await aiCacheGet(env, cacheKey); - if (pageTables) { - trackAi(env, { purpose: "extraction", model, cacheHit: true }); - } - else { - if (!(await limitAi(env))) - continue; - const t0 = Date.now(); - try { - const res = (await runAiWithTimeout(env.AI.run(model, { - messages: [ - { role: "system", content: "You extract tabular data from a single PDF page. Return strict JSON {tables: [{headers, rows}]}. Skip page numbers, app chrome (File/Edit/View toolbars, sheet tab strips, Share buttons), and prose paragraphs. If the page has no table, return {tables: []}. Each row must have the same length as headers; pad with empty strings if needed." }, - { role: "user", content: `PDF page text:\n${text}` }, - ], - response_format: { type: "json_schema", json_schema: TABLE_SCHEMA }, - }), AI_TIMEOUT_MS, "extract_tables")); - pageTables = parseTablesResponse(res); - } - catch (e) { - console.warn("aiExtractTablesFromPdfPages failed", e.message); - pageTables = []; - } - trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); - await aiCachePut(env, cacheKey, pageTables); - } - for (const t of pageTables) { - if (!Array.isArray(t.headers) || t.headers.length < 2) - continue; - if (!Array.isArray(t.rows) || t.rows.length < 1) - continue; - const headers = t.headers.map((h) => String(h || "").trim()); - const headerKey = headers.join("|").toLowerCase(); - const rows = t.rows.map((r) => { - const obj = {}; - for (let c = 0; c < headers.length; c++) - obj[headers[c] || `col_${c}`] = String(r?.[c] ?? "").trim(); - return obj; - }).filter((r) => Object.values(r).some((v) => v.length > 0)); - if (!rows.length) - continue; - if (lastHeaderKey === headerKey && out.length) { - out[out.length - 1].rows.push(...rows); - } - else { - out.push({ headers, rows, pageNumber: p + 1, confidence: 0.5 }); - lastHeaderKey = headerKey; - } - } - } - return out; -} -function parseTablesResponse(res) { - const r = res; - if (Array.isArray(r?.tables)) - return r.tables; - if (typeof r?.response === "string") { - try { - const j = JSON.parse(r.response); - if (Array.isArray(j?.tables)) - return j.tables; - } - catch { /* fall through */ } - } - return []; -} -export async function aiEmbed(env, text) { - if (!env.AI) - return null; - const model = env.AI_EMBED_MODEL ?? "@cf/baai/bge-base-en-v1.5"; - const cacheKey = await sha256Hex(`${model}:embed:${text}`); - const cached = await aiCacheGet(env, cacheKey); - if (cached) { - trackAi(env, { purpose: "embedding", model, cacheHit: true }); - return cached; - } - const ok = await assertBudget(env, "ai"); - if (!ok.ok) - return null; - if (!(await limitAi(env))) - return null; - const t0 = Date.now(); - try { - const res = (await runAiWithTimeout(env.AI.run(model, { text: [text] }), AI_TIMEOUT_SHORT_MS, "embed")); - const vec = Array.isArray(res?.data?.[0]) ? res.data[0] : null; - if (!vec) - return null; - trackAi(env, { purpose: "embedding", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); - await aiCachePut(env, cacheKey, vec); - return vec; - } - catch (e) { - console.warn("aiEmbed failed", e.message); - return null; - } -} -export async function aiArbitrate(env, candidateA, candidateB) { - if (!env.AI) - return { match: "maybe", confidence: 0 }; - const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; - const cacheKey = await sha256Hex(`${model}:arb:${candidateA}|${candidateB}`); - const cached = await aiCacheGet(env, cacheKey); - if (cached) { - trackAi(env, { purpose: "arbitration", model, cacheHit: true }); - return cached; - } - const ok = await assertBudget(env, "ai"); - if (!ok.ok) - return { match: "maybe", confidence: 0 }; - if (!(await limitAi(env))) - return { match: "maybe", confidence: 0 }; - const t0 = Date.now(); - try { - const res = (await runAiWithTimeout(env.AI.run(model, { - messages: [ - { role: "system", content: "Decide if two profiles describe the same person. Reply JSON: {match: yes|no|maybe, confidence: 0..1}." }, - { role: "user", content: `A: ${candidateA}\nB: ${candidateB}` }, - ], - response_format: { type: "json_object" }, - }), AI_TIMEOUT_SHORT_MS, "arbitrate")); - const out = parseArbResponse(res); - trackAi(env, { purpose: "arbitration", model, ms: Date.now() - t0, neurons: estimateNeurons(candidateA.length + candidateB.length) }); - await aiCachePut(env, cacheKey, out); - return out; - } - catch (e) { - console.warn("aiArbitrate failed", e.message); - return { match: "maybe", confidence: 0 }; - } -} -function parseArbResponse(res) { - if (typeof res?.response === "string") { - try { - const j = JSON.parse(res.response); - const match = j.match === "yes" || j.match === "no" ? j.match : "maybe"; - return { match, confidence: Math.max(0, Math.min(1, Number(j.confidence ?? 0))) }; - } - catch { /* fall through */ } - } - return { match: "maybe", confidence: 0 }; -} diff --git a/apps/worker/test-dist-q/analytics/events.js b/apps/worker/test-dist-q/analytics/events.js deleted file mode 100644 index c9fb991c..00000000 --- a/apps/worker/test-dist-q/analytics/events.js +++ /dev/null @@ -1,79 +0,0 @@ -// Analytics Engine event helpers (Task #25 step 7). -// -// Cloudflare Analytics Engine accepts up to 20 blobs (strings), 20 doubles, -// and 1 list of indexes per data point. We standardize the schema across all -// AI/fetch events so the GraphQL Analytics API queries stay simple. -// -// Schema: -// indexes: [purpose] e.g. "extraction" | "embedding" | "arbitration" | … -// blobs: [purpose, model, host, cache_hit, job_id] -// doubles: [neurons, ms, bytes, cost_usd] -export function trackAi(env, args) { - const purpose = args.purpose; - const cache = args.cacheHit ? "1" : "0"; - try { - env.ANALYTICS?.writeDataPoint({ - indexes: [purpose], - blobs: [purpose, args.model, "", cache, args.jobId ?? ""], - doubles: [args.neurons ?? 0, args.ms ?? 0, 0, args.costUsd ?? 0], - }); - } - catch { - /* analytics is best-effort */ - } - // Also persist a daily roll-up to D1 so /api/analytics/ae/ai-cost works - // without the GraphQL Analytics API (which requires an account-level token). - if (!args.cacheHit) { - void rollupAiCost(env, purpose, args.model, args.neurons ?? 0, args.costUsd ?? 0); - } -} -// Vectorize ops are billed per query/upsert independent of AI neurons. We -// count them in the same ai_cost_daily roll-up under purpose='vectorize_' -// so /api/scrapers/health can surface the daily burn against -// VECTORIZE_DAILY_QUERIES_CAP. -export function trackVectorize(env, args) { - try { - env.ANALYTICS?.writeDataPoint({ - indexes: [`vectorize_${args.op}`], - blobs: [`vectorize_${args.op}`, args.index, "", "0", ""], - doubles: [0, 0, 0, 0], - }); - } - catch { /* best-effort */ } - void rollupVectorizeOp(env, args.op, args.index); -} -async function rollupVectorizeOp(env, op, index) { - try { - const day = new Date().toISOString().slice(0, 10); - await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) - VALUES (?, ?, ?, 0, 0, 1) - ON CONFLICT(day, purpose, model) DO UPDATE SET calls = calls + 1`).bind(day, `vectorize_${op}`, index).run(); - } - catch (e) { - console.warn("vectorize rollup failed", e.message); - } -} -export function trackFetch(env, args) { - try { - env.ANALYTICS?.writeDataPoint({ - indexes: ["fetch"], - blobs: ["fetch", String(args.tier), args.host, args.blockReason ?? "", String(args.status)], - doubles: [0, args.ms, args.bytes, 0], - }); - } - catch { /* best-effort */ } -} -async function rollupAiCost(env, purpose, model, neurons, costUsd) { - try { - const day = new Date().toISOString().slice(0, 10); - await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) - VALUES (?, ?, ?, ?, ?, 1) - ON CONFLICT(day, purpose, model) DO UPDATE SET - neurons = neurons + excluded.neurons, - cost_usd = cost_usd + excluded.cost_usd, - calls = calls + 1`).bind(day, purpose, model, neurons, costUsd).run(); - } - catch (e) { - console.warn("ai_cost_daily rollup failed", e.message); - } -} diff --git a/apps/worker/test-dist-q/db/error_log.js b/apps/worker/test-dist-q/db/error_log.js deleted file mode 100644 index d4ae5fec..00000000 --- a/apps/worker/test-dist-q/db/error_log.js +++ /dev/null @@ -1,125 +0,0 @@ -// Task #27: structured error logging into D1 + Analytics Engine mirror. -// -// Best-effort writes — we never let a logging failure mask the original -// error. Truncation rules keep us under D1's 1MB row cap. Every call also -// emits one Analytics Engine data point so spike detection / alerting can -// be done without scanning D1. -import { AppError, wrapUnknown } from "../errors"; -const MAX_MSG = 2000; -const MAX_STACK = 8000; -const MAX_CONTEXT_JSON = 16000; -function clip(s, max) { - if (!s) - return null; - return s.length > max ? s.slice(0, max) + "…[clipped]" : s; -} -function hostFromUrl(u) { - if (!u) - return null; - try { - return new URL(u).hostname.toLowerCase(); - } - catch { - return null; - } -} -export async function logError(env, input) { - const e = input.err instanceof AppError ? input.err : wrapUnknown(input.err, "internal_error"); - const host = input.host ?? hostFromUrl(input.url); - // Mirror to Analytics Engine first (cheap, doesn't depend on D1). - try { - if (env.ANALYTICS) { - env.ANALYTICS.writeDataPoint({ - indexes: [e.code], - blobs: [ - e.kind, - e.code, - input.step ?? "", - input.job_id ?? "", - input.request_id ?? "", - host ?? "", - input.workflow_run_id ?? "", - input.user_email ?? "", - ], - doubles: [e.status, e.retryable ? 1 : 0, input.retry_count ?? 0], - }); - } - } - catch { /* never throw from logger */ } - if (!env.DB) - return null; - let contextJson = null; - try { - contextJson = clip(JSON.stringify(e.context ?? {}), MAX_CONTEXT_JSON); - } - catch { - contextJson = null; - } - try { - const r = await env.DB.prepare(`INSERT INTO error_log - (request_id, job_id, step, code, kind, status, retryable, message, context_json, - cause_name, cause_message, cause_stack, url, method, - workflow_run_id, host, user_email, retry_count) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) - .bind(input.request_id ?? null, input.job_id ?? null, input.step ?? null, e.code, e.kind, e.status, e.retryable ? 1 : 0, clip(e.message, MAX_MSG), contextJson, e.cause?.name ?? null, clip(e.cause?.message, MAX_MSG), clip(e.cause?.stack, MAX_STACK), input.url ?? null, input.method ?? null, input.workflow_run_id ?? null, host, input.user_email ?? null, input.retry_count ?? 0) - .run(); - const id = r.meta?.last_row_id; - return typeof id === "number" ? id : null; - } - catch (logErr) { - // Never throw from the logger. - // Telemetry-of-telemetry: never throw, never log to console (CI gate). - void logErr; - return null; - } -} -export async function logStep(env, input) { - if (!env.DB) - return; - let metaJson = null; - try { - metaJson = input.meta ? JSON.stringify(input.meta) : null; - } - catch { - metaJson = null; - } - const finishedAt = input.status === "started" ? null : new Date().toISOString(); - try { - await env.DB.prepare(`INSERT INTO workflow_step_log - (job_id, step, step_name, status, finished_at, duration_ms, count_in, count_out, error_id, meta_json, workflow_run_id, attempt, error_code) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) - .bind(input.job_id, input.step, input.step, input.status, finishedAt, input.duration_ms ?? null, input.count_in ?? null, input.count_out ?? null, input.error_id ?? null, metaJson, input.workflow_run_id ?? null, input.attempt ?? 1, input.error_code ?? null) - .run(); - } - catch (e) { - void e; - } -} -/** Convenience: time a function and log start+finish to workflow_step_log. */ -export async function timedStep(env, job_id, step, fn, opts = {}) { - const t0 = Date.now(); - const { workflow_run_id, attempt } = opts; - await logStep(env, { job_id, step, status: "started", count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt }); - try { - const out = await fn(); - const count_out = Array.isArray(out) ? out.length : undefined; - await logStep(env, { job_id, step, status: "ok", duration_ms: Date.now() - t0, count_in: opts.count_in, count_out, meta: opts.meta, workflow_run_id, attempt }); - return out; - } - catch (e) { - const { classify, isBenignSkip } = await import("../errors.js"); - // Task #72: robots.txt / ToS blocks are benign policy skips, not errors. - // Don't write an error_log row (it would surface as a red 422 in the - // operator console) — record the step as `skipped` instead and rethrow so - // the caller routes the job to the `skipped` terminal status. - const skip = isBenignSkip(e); - if (skip) { - await logStep(env, { job_id, step, status: "skipped", duration_ms: Date.now() - t0, count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt, error_code: skip.skip_code }); - throw e; - } - const error_id = await logError(env, { err: e, job_id, step }); - const cls = classify(e); - await logStep(env, { job_id, step, status: "error", duration_ms: Date.now() - t0, count_in: opts.count_in, error_id, meta: opts.meta, workflow_run_id, attempt, error_code: cls?.code }); - throw e; - } -} diff --git a/apps/worker/test-dist-q/entities/channels.js b/apps/worker/test-dist-q/entities/channels.js deleted file mode 100644 index 41a13c4c..00000000 --- a/apps/worker/test-dist-q/entities/channels.js +++ /dev/null @@ -1,61 +0,0 @@ -// Channel upsert keyed by (entity_id, kind, canonical). Canonical form -// is computed by ./normalize so trivially-different inputs collapse. -import { canonicalEmail, canonicalPhone, canonicalLinkedin, canonicalTwitter, canonicalGithub, canonicalUrl, } from "./normalize"; -export function canonicalizeFor(kind, raw) { - switch (kind) { - case "email": return canonicalEmail(raw); - case "phone": return canonicalPhone(raw); - case "linkedin": return canonicalLinkedin(raw); - case "twitter": return canonicalTwitter(raw); - case "github": return canonicalGithub(raw); - case "website": - case "other": - return canonicalUrl(raw); - } -} -export async function upsertChannel(env, input) { - if (!input.entity_id || !input.canonical) - return null; - // Re-canonicalize defensively in case caller passed a raw value. - const canonical = canonicalizeFor(input.kind, input.canonical) ?? input.canonical; - const now = new Date().toISOString(); - const existing = await env.DB.prepare(`SELECT id FROM channels WHERE entity_id = ? AND kind = ? AND canonical = ?`).bind(input.entity_id, input.kind, canonical).first(); - if (existing) { - const sets = ["last_seen_at = ?"]; - const binds = [now]; - if (input.display) { - sets.push("display = COALESCE(display, ?)"); - binds.push(input.display); - } - if (input.is_primary) { - sets.push("is_primary = 1"); - } - if (input.is_verified) { - sets.push("is_verified = 1"); - } - if (input.is_dnc) { - sets.push("is_dnc = 1"); - } - if (input.source) { - sets.push("source = COALESCE(source, ?)"); - binds.push(input.source); - } - binds.push(existing.id); - await env.DB.prepare(`UPDATE channels SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); - return existing.id; - } - const id = crypto.randomUUID(); - await env.DB.prepare(`INSERT INTO channels (id, entity_id, kind, canonical, display, is_primary, is_verified, is_dnc, source, confidence, first_seen_at, last_seen_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`).bind(id, input.entity_id, input.kind, canonical, input.display ?? null, input.is_primary ? 1 : 0, input.is_verified ? 1 : 0, input.is_dnc ? 1 : 0, input.source ?? null, input.confidence ?? 1, now, now).run(); - return id; -} -export async function findEntityByChannel(env, kind, raw) { - const canonical = canonicalizeFor(kind, raw); - if (!canonical) - return null; - const r = await env.DB.prepare(`SELECT c.entity_id FROM channels c - JOIN u_entities e ON e.id = c.entity_id - WHERE c.kind = ? AND c.canonical = ? AND e.status NOT IN ('merged','soft_deleted') - ORDER BY c.is_primary DESC, c.is_verified DESC, c.last_seen_at DESC LIMIT 1`).bind(kind, canonical).first(); - return r?.entity_id ?? null; -} diff --git a/apps/worker/test-dist-q/entities/dualwrite.js b/apps/worker/test-dist-q/entities/dualwrite.js deleted file mode 100644 index 505a51bc..00000000 --- a/apps/worker/test-dist-q/entities/dualwrite.js +++ /dev/null @@ -1,375 +0,0 @@ -// Dual-write hooks. Every legacy writer calls one of these to mirror the -// row into the unified entity model. All hooks are best-effort: they -// log + swallow errors so a transient unified-model failure never blocks -// the legacy path. -import { createEntity, addRole, getLegacyEntityId, setLegacyEntityId } from "./roles"; -import { insertFactsBatch } from "./facts"; -import { upsertChannel, findEntityByChannel } from "./channels"; -import { addTag, addTagsFromJsonArray } from "./tags"; -import { canonicalEmail, canonicalLinkedin, canonicalDomain } from "./normalize"; -import { enqueueSummaryRebuild } from "./summaryQueue"; -async function resolveOrCreate(env, table, legacyId, kind, init, channelLookups) { - const existing = await getLegacyEntityId(env, table, legacyId); - if (existing) - return existing; - // Cross-link by strongest available identifier before creating new: - // (a) deterministic domain match against u_entities.primary_domain - // (orgs only — collapses firm/account/company duplicates sharing a - // domain); (b) primary_email_key/primary_linkedin_key direct hits; - // (c) channel-table reverse lookups for any other handle. - if (kind === "org" && init.primary_domain) { - const r = await env.DB.prepare(`SELECT id FROM u_entities - WHERE primary_domain = ? AND status NOT IN ('merged','soft_deleted') - LIMIT 1`).bind(init.primary_domain).first(); - if (r?.id) { - await setLegacyEntityId(env, table, legacyId, r.id); - return r.id; - } - } - if (kind === "person" && init.primary_email_key) { - const r = await env.DB.prepare(`SELECT id FROM u_entities - WHERE primary_email_key = ? AND status NOT IN ('merged','soft_deleted') - LIMIT 1`).bind(init.primary_email_key).first(); - if (r?.id) { - await setLegacyEntityId(env, table, legacyId, r.id); - return r.id; - } - } - if (init.primary_linkedin_key) { - const r = await env.DB.prepare(`SELECT id FROM u_entities - WHERE primary_linkedin_key = ? AND status NOT IN ('merged','soft_deleted') - LIMIT 1`).bind(init.primary_linkedin_key).first(); - if (r?.id) { - await setLegacyEntityId(env, table, legacyId, r.id); - return r.id; - } - } - for (const ch of channelLookups) { - if (!ch.raw) - continue; - const hit = await findEntityByChannel(env, ch.kind, ch.raw); - if (hit) { - await setLegacyEntityId(env, table, legacyId, hit); - return hit; - } - } - try { - const created = await createEntity(env, { kind, ...init }); - if (!created) - return null; // Task #9: rejected by garbage detector - await setLegacyEntityId(env, table, legacyId, created.id); - return created.id; - } - catch (e) { - console.warn("dualwrite createEntity failed", table, legacyId, e.message); - return null; - } -} -function chan(env, entityId, kind, raw, source, primary = false) { - if (!raw) - return Promise.resolve(null); - return upsertChannel(env, { entity_id: entityId, kind, canonical: String(raw), source, is_primary: primary }); -} -export async function syncFirmToEntity(env, f, source = "firms_upsert", sourceKind = "scrape") { - try { - const domain = canonicalDomain(f.domain ?? f.website); - const linkedin = canonicalLinkedin(f.linkedin_url); - const entityId = await resolveOrCreate(env, "firms", f.id, "org", { - display_name: f.name, - primary_domain: domain, - primary_url: f.website ?? null, - primary_linkedin_key: linkedin, - }, [ - { kind: "linkedin", raw: f.linkedin_url }, - { kind: "website", raw: f.website ?? null }, - ]); - if (!entityId) - return null; - await addRole(env, entityId, "firm", { is_primary: true, source }); - if (f.kind && /accelerator/i.test(f.kind)) - await addRole(env, entityId, "accelerator", { source }); - // Task #2: cascading role inference on the unified write path so - // investor_firm / customer / prospect get assigned without each - // call-site having to know. - try { - const { inferAndAssignRoles } = await import("../services/roleInference.js"); - await inferAndAssignRoles(env, entityId, { - kind: "org", - sourceKind, - sourceUrl: f.website ?? null, - sourceDomain: domain ?? null, - org: f.name ?? null, - category: f.kind ?? null, - importLabel: source, - }); - } - catch (e) { - console.warn("inferAndAssignRoles(firm) failed", entityId, e.message); - } - const patches = [ - { predicate: "name", value_text: f.name }, - { predicate: "legal_name", value_text: f.legal_name ?? null }, - { predicate: "domain", value_text: domain }, - { predicate: "website", value_text: f.website ?? null }, - { predicate: "country_iso2", value_text: f.hq_country_iso2 ?? null }, - { predicate: "region", value_text: f.hq_region ?? null }, - { predicate: "city", value_text: f.hq_city ?? null }, - { predicate: "thesis", value_text: f.thesis ?? null }, - { predicate: "kind", value_text: f.kind ?? null }, - { predicate: "check_size_min_usd", value_number: numOrNull(f.check_size_min_usd) }, - { predicate: "check_size_max_usd", value_number: numOrNull(f.check_size_max_usd) }, - { predicate: "check_size_typical_usd", value_number: numOrNull(f.check_size_typical_usd) }, - ]; - await insertFactsBatch(env, entityId, patches, source, sourceKind); - await Promise.all([ - chan(env, entityId, "website", f.website, source, true), - chan(env, entityId, "linkedin", f.linkedin_url, source, true), - chan(env, entityId, "other", f.crunchbase_url, source), - chan(env, entityId, "twitter", f.twitter_handle, source), - chan(env, entityId, "email", f.contact_email, source), - ]); - await Promise.all([ - addTagsFromJsonArray(env, entityId, "sector", f.sectors_json, source), - addTagsFromJsonArray(env, entityId, "stage", f.stages_json, source), - addTagsFromJsonArray(env, entityId, "geo", f.geo_focus_json, source), - ]); - await enqueueSummaryRebuild(env, entityId); - return entityId; - } - catch (e) { - console.warn("syncFirmToEntity failed", f.id, e.message); - return null; - } -} -export async function syncLeadToEntity(env, l, source = "leads_repo", sourceKind = "scrape") { - try { - const email = canonicalEmail(l.email); - const linkedin = canonicalLinkedin(l.linkedin_url); - const entityId = await resolveOrCreate(env, "leads", l.id, "person", { - display_name: l.name ?? null, - primary_email_key: email, - primary_linkedin_key: linkedin, - primary_url: l.personal_url ?? null, - }, [ - { kind: "email", raw: l.email }, - { kind: "linkedin", raw: l.linkedin_url }, - ]); - if (!entityId) - return null; - // Role inference: investor_kind set → investor; otherwise generic person. - if (l.investor_kind) - await addRole(env, entityId, "investor", { is_primary: true, source }); - if (l.category && /founder|ceo|cto/i.test(l.category)) - await addRole(env, entityId, "founder", { source }); - // Task #2: cascading role inference on the unified write path. Runs - // for every person sync so the Investors page (which reads from - // entity_roles) picks up freshly-ingested people. Looks up the - // person's company entity for partner-title inheritance. - try { - const { inferAndAssignRoles } = await import("../services/roleInference.js"); - await inferAndAssignRoles(env, entityId, { - kind: "person", - sourceKind, - sourceUrl: l.source_url ?? l.personal_url ?? null, - sourceDomain: l.source_domain ?? null, - title: l.title ?? null, - org: l.org ?? null, - category: l.category ?? null, - importLabel: source, - }); - } - catch (e) { - console.warn("inferAndAssignRoles(lead) failed", entityId, e.message); - } - const patches = [ - { predicate: "name", value_text: l.name ?? null }, - { predicate: "title", value_text: l.title ?? null }, - { predicate: "primary_employer", value_text: l.org ?? null }, - { predicate: "category", value_text: l.category ?? null }, - { predicate: "country_iso2", value_text: l.country_iso2 ?? null }, - { predicate: "region", value_text: l.region ?? null }, - { predicate: "city", value_text: l.city ?? null }, - { predicate: "bio", value_text: l.bio ?? null }, - { predicate: "thesis", value_text: l.thesis ?? null }, - { predicate: "investor_kind", value_text: l.investor_kind ?? null }, - { predicate: "check_size_min_usd", value_number: numOrNull(l.check_size_min_usd) }, - { predicate: "check_size_max_usd", value_number: numOrNull(l.check_size_max_usd) }, - { predicate: "check_size_typical_usd", value_number: numOrNull(l.check_size_typical_usd) }, - ]; - await insertFactsBatch(env, entityId, patches, source, sourceKind); - await Promise.all([ - chan(env, entityId, "email", l.email, source, true), - chan(env, entityId, "phone", l.phone, source), - chan(env, entityId, "linkedin", l.linkedin_url, source, true), - chan(env, entityId, "twitter", l.twitter_url, source), - chan(env, entityId, "github", l.github_url, source), - chan(env, entityId, "website", l.personal_url, source), - ]); - await Promise.all([ - addTagsFromJsonArray(env, entityId, "sector", l.sector_focus_json, source), - addTagsFromJsonArray(env, entityId, "stage", l.stage_focus_json, source), - addTagsFromJsonArray(env, entityId, "geo", l.geo_focus_json, source), - addTagsFromJsonArray(env, entityId, "tag", l.tags_json, source), - ]); - await enqueueSummaryRebuild(env, entityId); - return entityId; - } - catch (e) { - console.warn("syncLeadToEntity failed", l.id, e.message); - return null; - } -} -export async function syncCompanyToEntity(env, c, source = "companies") { - try { - const domain = canonicalDomain(c.domain ?? c.website); - const linkedin = canonicalLinkedin(c.linkedin_url); - const entityId = await resolveOrCreate(env, "companies", c.id, "org", { - display_name: c.name, - primary_domain: domain, - primary_url: c.website ?? null, - primary_linkedin_key: linkedin, - }, [ - { kind: "linkedin", raw: c.linkedin_url }, - { kind: "website", raw: c.website ?? null }, - ]); - if (!entityId) - return null; - await addRole(env, entityId, "company", { is_primary: true, source }); - const patches = [ - { predicate: "name", value_text: c.name }, - { predicate: "legal_name", value_text: c.legal_name ?? null }, - { predicate: "domain", value_text: domain }, - { predicate: "website", value_text: c.website ?? null }, - { predicate: "country_iso2", value_text: c.hq_country_iso2 ?? null }, - { predicate: "region", value_text: c.hq_region ?? null }, - { predicate: "city", value_text: c.hq_city ?? null }, - { predicate: "stage", value_text: c.stage ?? null }, - { predicate: "unicorn_count", value_number: c.unicorn ? 1 : 0 }, - ]; - await insertFactsBatch(env, entityId, patches, source, "scrape"); - await Promise.all([ - chan(env, entityId, "website", c.website, source, true), - chan(env, entityId, "linkedin", c.linkedin_url, source), - chan(env, entityId, "twitter", c.twitter_handle, source), - chan(env, entityId, "github", c.github_org, source), - chan(env, entityId, "other", c.crunchbase_url, source), - ]); - await addTagsFromJsonArray(env, entityId, "sector", c.industries_json, source); - if (c.stage) - await addTag(env, { entity_id: entityId, taxonomy: "stage", slug: c.stage, source }); - await enqueueSummaryRebuild(env, entityId); - return entityId; - } - catch (e) { - console.warn("syncCompanyToEntity failed", c.id, e.message); - return null; - } -} -export async function syncAccountToEntity(env, a, source = "accounts") { - try { - const domain = canonicalDomain(a.domain ?? a.website); - const linkedin = canonicalLinkedin(a.linkedin_url); - const entityId = await resolveOrCreate(env, "accounts", a.id, "org", { - display_name: a.name, - primary_domain: domain, - primary_url: a.website ?? null, - primary_linkedin_key: linkedin, - }, [ - { kind: "linkedin", raw: a.linkedin_url }, - { kind: "website", raw: a.website ?? null }, - ]); - if (!entityId) - return null; - await addRole(env, entityId, "account", { is_primary: true, source }); - const patches = [ - { predicate: "name", value_text: a.name }, - { predicate: "legal_name", value_text: a.legal_name ?? null }, - { predicate: "domain", value_text: domain }, - { predicate: "website", value_text: a.website ?? null }, - { predicate: "country_iso2", value_text: a.hq_country_iso2 ?? null }, - { predicate: "region", value_text: a.hq_region ?? null }, - { predicate: "city", value_text: a.hq_city ?? null }, - { predicate: "industry", value_text: a.industry ?? null }, - { predicate: "funding_stage", value_text: a.funding_stage ?? null }, - { predicate: "fit_max_score", value_number: numOrNull(a.fit_score) }, - { predicate: "intent_score", value_number: numOrNull(a.intent_score) }, - // `accounts.employees` was the one sizing column dualwrite dropped, so - // no account entity carried a headcount fact and persona matching - // scored every one of them "company size unknown". `employees` is the - // predicate the registry declares (entities/profile-predicates.ts) and - // the one secEdgar/persist.ts already writes, so this converges on the - // name that exists rather than adding a fourth spelling. - { predicate: "employees", value_number: numOrNull(a.employees) }, - ]; - await insertFactsBatch(env, entityId, patches, source, "scrape"); - await Promise.all([ - chan(env, entityId, "website", a.website, source, true), - chan(env, entityId, "linkedin", a.linkedin_url, source), - chan(env, entityId, "twitter", a.twitter_handle, source), - chan(env, entityId, "github", a.github_org, source), - chan(env, entityId, "other", a.crunchbase_url, source), - ]); - await addTagsFromJsonArray(env, entityId, "sector", a.industries_json, source); - if (a.industry) - await addTag(env, { entity_id: entityId, taxonomy: "sector", slug: a.industry, source }); - await enqueueSummaryRebuild(env, entityId); - return entityId; - } - catch (e) { - console.warn("syncAccountToEntity failed", a.id, e.message); - return null; - } -} -export async function syncBuyerToEntity(env, b, source = "buyers") { - try { - const email = canonicalEmail(b.email); - const linkedin = canonicalLinkedin(b.linkedin_url); - const entityId = await resolveOrCreate(env, "buyers", b.id, "person", { - display_name: b.name ?? null, - primary_email_key: email, - primary_linkedin_key: linkedin, - }, [ - { kind: "email", raw: b.email }, - { kind: "linkedin", raw: b.linkedin_url }, - ]); - if (!entityId) - return null; - await addRole(env, entityId, "buyer", { is_primary: true, source }); - if (b.is_decision_maker) - await addRole(env, entityId, "executive", { source }); - // Link buyer → account via 'works_at' edge. - const accountEntityId = await getLegacyEntityId(env, "accounts", b.account_id); - if (accountEntityId) { - await env.DB.prepare(`INSERT INTO rel_edges (id, src_entity_id, dst_entity_id, kind, source) - VALUES (?, ?, ?, 'works_at', ?) - ON CONFLICT(src_entity_id, dst_entity_id, kind, IFNULL(valid_from,'')) DO NOTHING`).bind(crypto.randomUUID(), entityId, accountEntityId, source).run().catch(() => undefined); - // Also write as a fact for summary.primary_employer_entity_id pickup. - await insertFactsBatch(env, entityId, [{ predicate: "employer", value_entity_id: accountEntityId }], source, "inferred"); - } - const patches = [ - { predicate: "name", value_text: b.name ?? null }, - { predicate: "title", value_text: b.title ?? null }, - { predicate: "seniority", value_text: b.seniority ?? null }, - { predicate: "department", value_text: b.department ?? null }, - { predicate: "role_slug", value_text: b.role_slug ?? null }, - ]; - await insertFactsBatch(env, entityId, patches, source, "scrape"); - await Promise.all([ - chan(env, entityId, "email", b.email, source, true), - chan(env, entityId, "linkedin", b.linkedin_url, source, true), - chan(env, entityId, "twitter", b.twitter_url, source), - chan(env, entityId, "phone", b.phone, source), - ]); - if (b.role_slug) - await addTag(env, { entity_id: entityId, taxonomy: "role", slug: b.role_slug, source }); - await enqueueSummaryRebuild(env, entityId); - return entityId; - } - catch (e) { - console.warn("syncBuyerToEntity failed", b.id, e.message); - return null; - } -} -function numOrNull(v) { - return typeof v === "number" && Number.isFinite(v) ? v : null; -} diff --git a/apps/worker/test-dist-q/entities/facts.js b/apps/worker/test-dist-q/entities/facts.js deleted file mode 100644 index a40ff5df..00000000 --- a/apps/worker/test-dist-q/entities/facts.js +++ /dev/null @@ -1,282 +0,0 @@ -// Fact insertion with content-addressed dedup. The DB trigger -// `trg_facts_supersede` flips the prior is_current=1 fact for the same -// (entity, predicate, source) to 0 after insert. -import { sha256 } from "./normalize"; -import { enqueueSummaryRebuild } from "./summaryQueue"; -// Task #8: triggers debounced persona ↔ entity re-match when a fact -// that materially affects scoring is written. No-op otherwise. -import { triggerEntityMatchRefresh, isRelevantPredicate } from "../services/personaMatchTrigger"; -// Task #51: route previously-swallowed fact-write-path failures through the -// structured error logger so they land in `error_log` instead of vanishing. -import { logError } from "../db/error_log"; -export async function insertFact(env, f) { - if (!f.entity_id || !f.predicate) - return null; - const valueKey = JSON.stringify({ - t: f.value_text ?? null, - n: f.value_number ?? null, - j: f.value_json ?? null, - e: f.value_entity_id ?? null, - }); - const hash = await sha256(`${f.entity_id}|${f.predicate}|${valueKey}|${f.source ?? ""}`); - const id = crypto.randomUUID(); - const now = f.observed_at ?? new Date().toISOString(); - // Task #3 (Editable Profiles): lock check. If a locked override exists - // for this (entity, predicate), the new fact row is still inserted (so - // the diff strip can show the AI/scrape attempt) but stamped with - // superseded_by_override=1 so it never wins the read race. The override - // layer overlays at read time via getEffectiveFacts. - // Wrapped in try/catch (not just `.catch`) so a missing `field_overrides` - // table — fresh install ahead of migration 376, or a minimal test DB whose - // prepare() throws synchronously — is logged and degrades to "no lock" - // instead of failing the fact write. - const lock = await (async () => { - try { - return await env.DB.prepare(`SELECT 1 FROM field_overrides - WHERE entity_id = ? AND predicate = ? AND locked = 1 - AND (unlock_after IS NULL OR unlock_after > datetime('now')) - LIMIT 1`).bind(f.entity_id, f.predicate).first(); - } - catch (e) { - await logError(env, { err: e, step: "facts.insertFact.override_lock_check" }); - return null; - } - })(); - const supersededByOverride = lock ? 1 : 0; - try { - await env.DB.prepare(`INSERT INTO facts ( - id, entity_id, predicate, value_text, value_number, value_json, - value_entity_id, source_kind, source, evidence_url, confidence, - observed_at, valid_from, valid_to, is_current, hash, superseded_by_override - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 1, ?, ?)`).bind(id, f.entity_id, f.predicate, f.value_text ?? null, f.value_number ?? null, f.value_json != null ? JSON.stringify(f.value_json) : null, f.value_entity_id ?? null, f.source_kind, f.source ?? null, f.evidence_url ?? null, f.confidence ?? 1, now, f.valid_from ?? null, f.valid_to ?? null, hash, supersededByOverride).run(); - // Task #3 race fix: the SELECT lock-check above and this INSERT are - // not atomic. If an override landed between them, our row would have - // superseded_by_override=0 even though an override now dominates. The - // override-create handler ALSO runs `UPDATE facts SET - // superseded_by_override = 1` to catch facts inserted before the - // override; this post-insert re-check covers the reverse direction, - // so both writers converge on the same end state regardless of which - // raced first. - if (!supersededByOverride) { - await env.DB.prepare(`UPDATE facts SET superseded_by_override = 1 - WHERE id = ? - AND EXISTS ( - SELECT 1 FROM field_overrides - WHERE entity_id = ? AND predicate = ? AND locked = 1 - AND (unlock_after IS NULL OR unlock_after > datetime('now')) - )`).bind(id, f.entity_id, f.predicate).run().catch((e) => logError(env, { err: e, step: "facts.insertFact.override_recheck" })); - } - // Centralized rebuild guarantee: every successful fact insert - // enqueues a summary rebuild for the owning entity. This keeps the - // "fact INSERT → rebuild within ~5s" SLO honest regardless of which - // caller wrote the fact (dual-write, merge, manual admin, etc.). - await enqueueSummaryRebuild(env, f.entity_id); - if (isRelevantPredicate(f.predicate)) { - // Fire-and-forget; debounced via KV inside the trigger. - void triggerEntityMatchRefresh(env, f.entity_id).catch((e) => { - console.warn("triggerEntityMatchRefresh from insertFact failed", e.message); - }); - } - // Task #4 (Relationship Inference Worker): debounced enqueue into - // relationship_infer_queue (migration 377). KV-debounced 60s; the - // consolidated nightly slot drains the queue with the per-entity - // orchestrator pass. Never inline — entity/fact writes stay fast. - // No `relationship_infer` JobKind exists, so we fall back to the - // nightly tick per the spec's explicit instruction. - try { - const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); - void enqueueRelInfer(env, f.entity_id, `fact:${f.predicate}`).catch((e) => logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer" })); - } - catch (e) { - void logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer_import" }); - } - return id; - } - catch (e) { - // UNIQUE(hash) collision = exact-replay observation. Task #1 - // requires re-imports of the same Folk row to refresh `observed_at` - // so freshness queries reflect when we last *saw* the fact, even - // when nothing about the value changed. We update the existing row - // (matched by hash) instead of writing a new one. - const msg = e.message || ""; - if (/UNIQUE/i.test(msg)) { - try { - await env.DB.prepare("UPDATE facts SET observed_at = ? WHERE hash = ?").bind(now, hash).run(); - } - catch (uErr) { - console.warn("insertFact observed_at refresh failed", uErr.message); - } - return null; - } - throw e; - } -} -function parseJsonSafe(s) { - if (s == null) - return null; - try { - return JSON.parse(s); - } - catch { - return s; - } -} -export async function loadCurrentOverrides(env, entityId) { - const r = await env.DB.prepare(`SELECT id, predicate, value_text, value_numeric, value_json, overridden_at - FROM field_overrides - WHERE entity_id = ? AND locked = 1 - AND (unlock_after IS NULL OR unlock_after > datetime('now')) - ORDER BY overridden_at DESC`).bind(entityId).all().catch(() => ({ results: [] })); - const map = new Map(); - for (const o of r.results ?? []) { - if (!map.has(o.predicate)) - map.set(o.predicate, o); - } - return map; -} -export async function getEffectiveFacts(env, entityId, opts) { - const factWhere = opts?.includeNonCurrent ? "" : " AND is_current = 1"; - const limit = opts?.limit ?? 500; - const [factsRes, overrides] = await Promise.all([ - env.DB.prepare(`SELECT id, predicate, value_text, value_number, value_json, value_entity_id, - source_kind, source, confidence, verified_score, observed_at, - is_current, superseded_by_override - FROM facts - WHERE entity_id = ?${factWhere} - ORDER BY observed_at DESC LIMIT ?`).bind(entityId, limit).all(), - loadCurrentOverrides(env, entityId), - ]); - const out = []; - const overridePredsSeen = new Set(); - for (const f of factsRes.results ?? []) { - const ov = overrides.get(f.predicate); - if (ov) { - if (!overridePredsSeen.has(f.predicate)) { - overridePredsSeen.add(f.predicate); - out.push({ - id: `override:${ov.id}`, - predicate: f.predicate, - value_text: ov.value_text, - value_number: ov.value_numeric, - value_json: parseJsonSafe(ov.value_json), - value_entity_id: null, - source_kind: "manual", - source: "field_override", - confidence: 1, - verified_score: null, - observed_at: ov.overridden_at, - is_current: 1, - superseded_by_override: 0, - is_override: true, - override_id: ov.id, - overridden_attempt: false, - }); - } - // Mark the underlying fact as an overridden attempt for the diff - // strip. Never returned as canonical. - out.push({ - id: f.id, - predicate: f.predicate, - value_text: f.value_text, - value_number: f.value_number, - value_json: parseJsonSafe(f.value_json), - value_entity_id: f.value_entity_id, - source_kind: f.source_kind, - source: f.source, - confidence: f.confidence, - verified_score: f.verified_score, - observed_at: f.observed_at, - is_current: f.is_current, - superseded_by_override: 1, - is_override: false, - override_id: null, - overridden_attempt: true, - }); - } - else if (f.superseded_by_override === 1) { - out.push({ - id: f.id, - predicate: f.predicate, - value_text: f.value_text, - value_number: f.value_number, - value_json: parseJsonSafe(f.value_json), - value_entity_id: f.value_entity_id, - source_kind: f.source_kind, - source: f.source, - confidence: f.confidence, - verified_score: f.verified_score, - observed_at: f.observed_at, - is_current: f.is_current, - superseded_by_override: 1, - is_override: false, - override_id: null, - overridden_attempt: true, - }); - } - else { - out.push({ - id: f.id, - predicate: f.predicate, - value_text: f.value_text, - value_number: f.value_number, - value_json: parseJsonSafe(f.value_json), - value_entity_id: f.value_entity_id, - source_kind: f.source_kind, - source: f.source, - confidence: f.confidence, - verified_score: f.verified_score, - observed_at: f.observed_at, - is_current: f.is_current, - superseded_by_override: 0, - is_override: false, - override_id: null, - overridden_attempt: false, - }); - } - } - // Overrides for predicates with no underlying fact row at all. - for (const [pred, ov] of overrides.entries()) { - if (overridePredsSeen.has(pred)) - continue; - out.push({ - id: `override:${ov.id}`, - predicate: pred, - value_text: ov.value_text, - value_number: ov.value_numeric, - value_json: parseJsonSafe(ov.value_json), - value_entity_id: null, - source_kind: "manual", - source: "field_override", - confidence: 1, - verified_score: null, - observed_at: ov.overridden_at, - is_current: 1, - superseded_by_override: 0, - is_override: true, - override_id: ov.id, - overridden_attempt: false, - }); - } - return out; -} -export async function insertFactsBatch(env, entityId, patches, source, sourceKind = "scrape", evidenceUrl = null) { - let n = 0; - for (const p of patches) { - if (p.value_text == null && p.value_number == null && p.value_json == null && p.value_entity_id == null) - continue; - const id = await insertFact(env, { - entity_id: entityId, - predicate: p.predicate, - value_text: p.value_text ?? null, - value_number: p.value_number ?? null, - value_json: p.value_json, - value_entity_id: p.value_entity_id ?? null, - source_kind: sourceKind, - source, - evidence_url: evidenceUrl, - }); - if (id) - n += 1; - } - return n; -} diff --git a/apps/worker/test-dist-q/entities/garbage.js b/apps/worker/test-dist-q/entities/garbage.js deleted file mode 100644 index ffa95393..00000000 --- a/apps/worker/test-dist-q/entities/garbage.js +++ /dev/null @@ -1,652 +0,0 @@ -// Task #9: Garbage Entity Detector & Cleanup. -// -// Pure detector (`isGarbage`) flags HTML page titles / nav fragments / -// UI strings polluting `u_entities`. Used by: -// * The pre-insert guard in `createEntity` (rejects before write). -// * The cron sweep (`runCleanupSweep`) that soft-deletes recently- -// created garbage and, on `mode='all'`, performs the one-off pass. -// * The /ops/garbage-review/ console (admin restore / purge). -// -// HONEST DEGRADATION (Task #14 pattern): the optional Workers AI second -// opinion (`aiSecondOpinion`) returns `uncertain` when the `env.AI` -// binding is absent, on any HTTP/network error, or when the JSON -// response is malformed. `evaluateEntity` then DOES NOT flag the -// entity — never silently garbage. -const NAME_MAX_LEN = 80; -// Curated UI / nav strings observed in production on the Investors page. -// Lowercased for comparison. -const KNOWN_UI_STRINGS = new Set([ - "contact us", "contact", "search icon", "search", "home", "about", - "about us", "menu", "login", "log in", "sign in", "sign up", - "sign-up", "register", "our team", "team", "limited partners", - "portfolio", "our portfolio", "careers", "jobs", "privacy", - "privacy policy", "terms", "terms of service", "cookies", - "cookie policy", "blog", "news", "press", "press releases", - "get in touch", "subscribe", "newsletter", "footer", "header", - "navigation", "nav", "skip to content", "back to top", "read more", - "learn more", "view all", "see all", "all rights reserved", - "follow us", "share", "tweet", "facebook", "twitter", "linkedin", - "instagram", "youtube", "the team", "our story", -]); -// Heuristic leaders that strongly indicate a press/blog post title -// got captured as an "entity". -const LEADER_RE = /^(introducing|announcing|welcome to|how|why|what|when|where|the future of|inside)\s+/i; -// Page-title with `|`-separated domain/brand fragment. Examples: -// "Our Team | Sequoia Capital", "Contact Tenity | Get in Touch", -// "Home | Sequoia Capital". -const PIPE_TITLE_RE = /\s\|\s\S/; -// Pure emoji / icon names (no alphanumerics at all). -const NO_ALNUM_RE = /^[^\p{L}\p{N}]+$/u; -// Listicle / directory page titles captured as entity names. -// -// This is the gap that let ~128 non-firms into the firms table: a crawler -// ingested aggregator pages ("VC Firms By Stage" on failory.com) and made -// one entity per outbound link, taking the page title as the name. The -// existing rules could not catch it — such a title has no pipe fragment, no -// press leader, is well under 80 characters and is not a known nav string. -// -// Deliberately narrow, because a single matched reason marks an entity -// garbage. Each pattern is a phrase a real firm name essentially never -// contains: "Top Tier Capital Partners" and "Stage Fund" are real firms, so -// a bare leading "top" or the word "stage" alone must NOT match. -const LISTICLE_RES = [ - // "VC Firms By Stage", "Investors by sector", "Funds per geography" - /\b(?:by|per)\s+(?:stage|sector|industry|geograph|countr|region|check\s*size|vertical)/i, - // "Top 50 VC Firms", "Best 10 Seed Funds" — the number is what makes this - // safe; a leading "Top"/"Best" alone is a legitimate name fragment. - /^(?:the\s+)?(?:top|best|leading)\s+\d+\b/i, - // "List of European VCs", "Directory of angel investors" - /\b(?:list|directory|database|roundup|ranking)\s+of\s+/i, - // "The Ultimate Guide to Seed Funds", "Complete List of ..." - /^(?:the\s+)?(?:complete|ultimate|definitive)\s+(?:list|guide|directory|database)\b/i, -]; -// Minimum length before a name that is identical to its own domain slug is -// treated as URL-derived rather than a genuine single-word brand. -// "Stripe" / "Coatue" / "Atomico" are real names that equal their domain; -// "Firstmarkcap" (firstmarkcap.com) and "Forerunnerventures" are slugs that -// were title-cased because no real name was ever extracted. -const SLUG_NAME_MIN_LEN = 12; -/** - * True when the display name is just the registrable domain label with the - * first letter capitalised — i.e. the crawler never found a name and fell - * back to the URL. Length-gated so short single-word brands are untouched. - */ -function looksDomainDerived(name, domain) { - if (!domain) - return false; - const label = domain.toLowerCase().replace(/^www\./, "").split(".")[0] ?? ""; - if (label.length < SLUG_NAME_MIN_LEN) - return false; - const n = name.trim().toLowerCase(); - // A genuine name carries separators the slug cannot ("First Mark Capital"). - if (/[\s.\-_]/.test(n)) - return false; - return n === label; -} -// --------------------------------------------------------------------------- -// Task #6: person-name disambiguation. Classifies a name that was recorded -// as a `person` into one of: a real person, an organization scraped as a -// person (firm / fund / accelerator / company), generic page junk, or -// uncertain. Used by: -// * the pre-insert reclassify-on-write guard in `createEntity`, -// * the cron / one-off sweep (`runCleanupSweep`), -// * the scraper extraction boundary (`extractPeopleFromPage`). -// PURE — name-only, no IO — so it's safe on the hot write path. -// --------------------------------------------------------------------------- -// Legal-entity suffixes — an extremely strong organization signal anywhere -// in the name. Normalized (punctuation stripped) before comparison. -const ORG_LEGAL_SUFFIX = new Set([ - "llc", "inc", "ltd", "limited", "lp", "llp", "plc", "gmbh", "ag", - "sarl", "bv", "pty", "oy", "ab", "srl", "spa", -]); -// Descriptor words that, as the LAST token, denote an organization -// ("Intel Capital", "Mendoza Ventures", "Hillman Accelerator Foundation"). -const ORG_SUFFIX_LAST = new Set([ - "capital", "ventures", "venture", "partners", "partner", "holdings", - "group", "fund", "funds", "foundation", "labs", "lab", "hub", - "collective", "management", "advisors", "associates", "accelerator", - "incubator", "equity", "securities", "technologies", "studios", "studio", - "network", "institute", "academy", "council", "alliance", "syndicate", - "consortium", "enterprises", "industries", "international", "global", - "company", "corp", "corporation", "university", "college", "systems", - "solutions", -]); -// Generic, non-distinctive words. A name made up ENTIRELY of these is junk -// ("Deep Tech", "Our Mission"); they're also excluded when looking for a -// distinctive proper-noun token in an org name. -const GENERIC_WORDS = new Set([ - "the", "our", "your", "my", "a", "an", "all", "more", "new", "updated", - "featured", "latest", "recent", "top", "best", "of", "and", "or", "for", - "with", "to", "in", "on", "at", "by", "from", "about", "welcome", "hello", - "home", "homepage", "page", "web", "website", "webpage", "mission", - "vision", "values", "story", "team", "careers", "jobs", "blog", "news", - "press", "media", "map", "menu", "footer", "header", "sidebar", "gallery", - "resources", "events", "podcast", "newsletter", "insights", "research", - "report", "reports", "overview", "summary", "services", "solutions", - "products", "pricing", "features", "get", "started", "learn", "read", - "view", "see", "guide", "guides", "faq", "faqs", "help", "support", - "contact", "deep", "tech", "technology", "startup", "startups", - "mentorship", "money", "data", "signal", "community", "ecosystem", - "platform", "world", "global", "international", "region", "regions", - "area", "areas", "north", "south", "east", "west", "central", "america", - "americas", "europe", "asia", "africa", "oceania", "antarctica", "middle", - "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", - "thumbnail", "placeholder", "avatar", "headshot", "slideshow", "carousel", - "machine", "wayback", "future", "work", "working", "people", "portfolio", - "companies", "investors", "founders", "funding", "rounds", "deals", -]); -// Words whose presence alone marks a name as page junk rather than a -// person — decorative / UI / asset captions that pass NAME_RE. -const HARD_JUNK_WORDS = new Set([ - "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", - "thumbnail", "placeholder", "gallery", "slideshow", "carousel", "homepage", - "webpage", "sidebar", "footer", "header", "menu", "map", "wayback", - "machine", -]); -// Exact lowercase phrases observed polluting the People list. -const KNOWN_JUNK_PHRASES = new Set([ - "updated homepage image", "our mission", "map of the money", - "wayback machine", "deep tech", "startup mentorship hub", "read more", - "learn more", "our team", "the team", "our story", "our values", - "our vision", "get started", "coming soon", "page not found", -]); -// Exact lowercase place / region names that get scraped as "people". -const PLACE_NAMES = new Set([ - "north america", "south america", "central america", "latin america", - "united states", "united kingdom", "middle east", "european union", - "north", "south", "east", "west", "europe", "asia", "africa", "oceania", - "antarctica", "americas", "global", "worldwide", -]); -function normToken(t) { - return t.toLowerCase().normalize("NFKD").replace(/[^a-z0-9]/g, ""); -} -function orgRoleForTokens(tokensLower) { - const set = new Set(tokensLower); - if (set.has("accelerator") || set.has("incubator")) - return "accelerator"; - if (set.has("fund") || set.has("funds")) - return "fund"; - for (const t of ["capital", "ventures", "venture", "partners", "partner", - "equity", "management", "advisors", "associates", "holdings", - "securities", "syndicate"]) { - if (set.has(t)) - return "investor_firm"; - } - return "firm"; -} -/** Map an inferred org role to the `firms.kind` taxonomy for dual-write. */ -export function orgRoleToFirmKind(role) { - switch (role) { - case "accelerator": return "accelerator"; - case "fund": return "fund"; - case "investor_firm": return "vc"; - default: return null; - } -} -/** - * Pure classifier for a name recorded as a `person`. Conservative by - * design: only returns `organization` / `junk` when the signal is clear, - * otherwise `person` (a plausible human name) or `uncertain`. Callers - * decide what to DO with each verdict (reclassify, soft-delete, review). - */ -export function classifyPersonName(rawName) { - const raw = (rawName ?? "").trim(); - if (!raw) - return { verdict: "junk", reasons: ["empty_name"] }; - const lower = raw.toLowerCase(); - const tokens = raw.split(/\s+/).filter(Boolean); - const norm = tokens.map(normToken).filter(Boolean); - // 1. Exact known-junk phrase / place name. - if (KNOWN_JUNK_PHRASES.has(lower)) - return { verdict: "junk", reasons: ["known_junk_phrase"] }; - if (PLACE_NAMES.has(lower)) - return { verdict: "junk", reasons: ["place_name"] }; - // 2. Hard junk word present (image / logo / homepage / map / ...). - // PRECISION GUARD (precision-over-recall): a single junk token inside an - // otherwise-clean two-token Title-Case name ("John Banner", "John Map") - // must NOT auto-delete a plausible real person. Real decorative captions - // are ≥3 tokens ("Updated Homepage Image") or all-generic two-token - // phrases ("Wayback Machine", caught by rule 4 below), so deferring the - // junk-word rule for clean two-token names keeps every junk fixture while - // protecting people whose surname happens to collide with an asset word. - const cleanTwoToken = tokens.length === 2 && tokens.every((t) => /^[\p{Lu}][\p{L}'’.\-]*$/u.test(t)); - if (!cleanTwoToken) { - for (const t of norm) { - if (HARD_JUNK_WORDS.has(t)) - return { verdict: "junk", reasons: [`junk_word:${t}`] }; - } - } - // 3. Organization-suffix detection. - const last = norm[norm.length - 1] ?? ""; - const hasLegal = norm.some((t) => ORG_LEGAL_SUFFIX.has(t)); - const lastIsOrgSuffix = ORG_SUFFIX_LAST.has(last); - if (hasLegal || lastIsOrgSuffix) { - // Need a distinctive (non-generic, non-suffix) token to call it a real - // org. "Intel Capital" → distinctive "intel". "Startup Mentorship Hub" - // → all-generic + suffix → junk. - const distinctive = norm.filter((t, i) => !GENERIC_WORDS.has(t) && - !ORG_SUFFIX_LAST.has(t) && - !ORG_LEGAL_SUFFIX.has(t) && - !(i === norm.length - 1 && lastIsOrgSuffix)); - if (distinctive.length === 0) { - return { verdict: "junk", reasons: ["generic_org_phrase"] }; - } - return { - verdict: "organization", - orgRole: orgRoleForTokens(norm), - reasons: [hasLegal ? "org_legal_suffix" : `org_suffix:${last}`], - }; - } - // 4. Every token is a generic word ("Deep Tech", "Our Mission"). - if (norm.length >= 1 && norm.every((t) => GENERIC_WORDS.has(t))) { - return { verdict: "junk", reasons: ["all_generic_words"] }; - } - // 5. Plausible human name: 2–4 tokens, ≥2 capitalized, not all generic. - const titleTokens = tokens.filter((t) => /^[\p{Lu}]/u.test(t)); - if (tokens.length >= 2 && tokens.length <= 4 && titleTokens.length >= 2) { - return { verdict: "person", reasons: ["plausible_person_name"] }; - } - // 6. Anything else — don't guess. - return { verdict: "uncertain", reasons: ["unclassified"] }; -} -/** Pure detector. NO IO. Safe to call inline on every entity write. */ -export function isGarbage(input) { - const reasons = []; - const raw = (input.display_name ?? "").trim(); - // Rule 1: empty or whitespace-only name. - if (!raw) { - reasons.push("empty_name"); - return { is_garbage: true, reasons }; - } - // Rule 2: name longer than 80 chars. - if (raw.length > NAME_MAX_LEN) - reasons.push("name_too_long"); - // Rule 3: pure emoji / icon (no letters or digits). - if (NO_ALNUM_RE.test(raw)) - reasons.push("no_alphanumeric_chars"); - // Rule 4: page-title with `|` brand fragment. - if (PIPE_TITLE_RE.test(raw)) - reasons.push("page_title_pipe_fragment"); - // Rule 5: blog/press leader phrase. - if (LEADER_RE.test(raw)) - reasons.push("press_leader_phrase"); - // Rule 6: known UI / nav string (case-insensitive exact match). - if (KNOWN_UI_STRINGS.has(raw.toLowerCase())) - reasons.push("known_ui_string"); - // Rule 6c: listicle / directory page title captured as an entity name. - if (LISTICLE_RES.some((re) => re.test(raw))) - reasons.push("listicle_page_title"); - // Rule 6d: name is just the domain slug — the crawler never extracted a - // real name and fell back to the URL. - if (looksDomainDerived(raw, input.primary_domain)) - reasons.push("domain_slug_name"); - // Rule 6b (Task #6 Section A/F): literal HTML entity in name - // (e.g. "Founder & Partner", "Acme & Co"). These are parser - // bugs upstream — the entity should have been decoded before write. - // We flag them here as garbage so they soft-delete on the next sweep - // and surface to the operator console; the durable fix is at the - // scraper layer via decodeEntities() (Task #6 Section F). - if (/&(amp|lt|gt|quot|#x?[0-9a-f]+);/i.test(raw)) - reasons.push("literal_html_entity"); - // Rule 7: person-specific constraints — must contain a space AND - // must not contain pipe / slash / colon. Real human display names - // are "First Last", not "Contact | Sequoia" or "team/people:1". - if (input.kind === "person") { - if (!/\s/.test(raw)) - reasons.push("person_no_space"); - if (/[|/:]/.test(raw)) - reasons.push("person_contains_separator"); - // Task #6: generic page-junk names recorded as people ("Updated - // Homepage Image", "Our Mission", "North America"). Organization - // names are NOT flagged here — they're reclassified (not deleted) - // by the createEntity write guard and the sweep. - const cls = classifyPersonName(raw); - if (cls.verdict === "junk") { - for (const code of cls.reasons) - reasons.push(`name_${code}`); - } - } - return { is_garbage: reasons.length > 0, reasons }; -} -// --------------------------------------------------------------------------- -// Structural rule (requires DB lookups): zero facts AND zero relationships -// AND zero contact channels AND crawler-created >24h ago. Used by the -// cron sweep — NOT by the pre-insert guard (the entity hasn't been -// written yet, so it has no joins). -// --------------------------------------------------------------------------- -export async function isStructurallyOrphan(env, entityId, options = {}) { - const minAge = options.minAgeHours ?? 24; - const reasons = []; - try { - const row = await env.DB.prepare(`SELECT - (SELECT COUNT(*) FROM facts WHERE entity_id = ?1) AS facts, - (SELECT COUNT(*) FROM rel_edges WHERE src_entity_id = ?1 OR dst_entity_id = ?1) AS rels, - (SELECT COUNT(*) FROM channels WHERE entity_id = ?1) AS chans, - (SELECT (julianday('now') - julianday(created_at)) * 24 FROM u_entities WHERE id = ?1) AS age_hours`).bind(entityId).first(); - if (!row) - return { orphan: false, reasons }; - const ageHours = Number(row.age_hours ?? 0); - if (Number(row.facts) === 0 && Number(row.rels) === 0 && Number(row.chans) === 0 && ageHours >= minAge) { - reasons.push("structural_orphan_no_signal"); - return { orphan: true, reasons }; - } - } - catch (e) { - // Optional source tables (channels) may be missing in test - // DBs — degrade to "not orphan" rather than throwing. Per the - // Task #14 honest-degradation pattern. - console.warn("isStructurallyOrphan probe failed", entityId, e.message); - } - return { orphan: false, reasons }; -} -const AI_PROMPT = `You are a data-quality filter for a CRM. Given a candidate \ -entity record, decide whether the display_name is a real person/organization \ -name or noise scraped from an HTML page (page titles, nav labels, "Contact Us", \ -press headlines like "Introducing X", marketing blurbs, etc.). -Reply ONLY as compact JSON: {"verdict":"garbage|real|uncertain","confidence":0.0-1.0,"reason":""}.`; -export async function aiSecondOpinion(env, input) { - if (!env.AI || typeof env.AI.run !== "function") { - return { verdict: "uncertain", confidence: 0, reason: "ai_binding_missing" }; - } - const payload = { - kind: input.kind, - display_name: input.display_name ?? null, - primary_url: input.primary_url ?? null, - primary_domain: input.primary_domain ?? null, - }; - try { - const res = (await env.AI.run("@cf/meta/llama-3.1-8b-instruct-fast", { - messages: [ - { role: "system", content: AI_PROMPT }, - { role: "user", content: JSON.stringify(payload) }, - ], - max_tokens: 80, - })); - const text = typeof res === "string" ? res : (res?.response ?? ""); - const match = text.match(/\{[\s\S]*\}/); - if (!match) - return { verdict: "uncertain", confidence: 0, reason: "ai_no_json" }; - const parsed = JSON.parse(match[0]); - const verdict = parsed.verdict === "garbage" || parsed.verdict === "real" ? parsed.verdict : "uncertain"; - const conf = typeof parsed.confidence === "number" && Number.isFinite(parsed.confidence) - ? Math.max(0, Math.min(1, parsed.confidence)) - : 0; - return { verdict, confidence: conf, reason: typeof parsed.reason === "string" ? parsed.reason : undefined }; - } - catch (e) { - return { verdict: "uncertain", confidence: 0, reason: "ai_error:" + e.message }; - } -} -/** - * Combined verdict: heuristic detector + optional AI second opinion - * for names 30–60 chars that don't match any heuristic. AI flags only - * when verdict='garbage' AND confidence > 0.8. When AI is unavailable - * or returns 'uncertain', the entity is NOT flagged. - */ -export async function evaluateEntity(env, input, opts = {}) { - const heur = isGarbage(input); - if (heur.is_garbage) - return heur; - if (opts.skipAi) - return heur; - const name = (input.display_name ?? "").trim(); - if (name.length < 30 || name.length > 60) - return heur; - const ai = await aiSecondOpinion(env, input); - if (ai.verdict === "garbage" && ai.confidence > 0.8) { - return { is_garbage: true, reasons: ["ai_second_opinion", `ai_conf:${ai.confidence.toFixed(2)}`] }; - } - return heur; -} -// --------------------------------------------------------------------------- -// Soft-delete / restore / purge helpers. All write through -// `data_quality_log` so the operator console can audit every transition. -// --------------------------------------------------------------------------- -export async function logDataQuality(env, entityId, issue, reasons, source, actorEmail, -/** Stored in reasons_json instead of `reasons` when given. Used to park a - * structured snapshot (see softDeleteEntity) rather than a reason list. */ -payload) { - try { - await env.DB.prepare(`INSERT INTO data_quality_log (entity_id, issue, reasons_json, source, actor_email) - VALUES (?, ?, ?, ?, ?)`).bind(entityId, issue, JSON.stringify(payload ?? reasons), source, actorEmail ?? null).run(); - } - catch (e) { - console.warn("data_quality_log insert failed", entityId, e.message); - } -} -export async function softDeleteEntity(env, entityId, reasons, source, actorEmail) { - await env.DB.prepare(`UPDATE u_entities - SET status = 'soft_deleted', - deleted_reason = COALESCE(deleted_reason, ?), - updated_at = datetime('now') - WHERE id = ? AND status != 'soft_deleted'`).bind("garbage_detector_v1:" + reasons.join(","), entityId).run(); - try { - // Park the roles before deleting them. Without this the delete is a - // one-way door: restoreEntity flips status back to 'active' and nothing - // puts the roles back, so a restored entity returns with no roles at all - // and stays invisible on every role-filtered surface — the investor - // lists, the persona matchers, the founder screens. That makes - // "quarantine, then restore the false positives" a promise the code - // cannot keep, which is the whole point of soft delete over purge. - const roles = await env.DB.prepare(`SELECT role, is_primary, source, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(); - // Parked unconditionally, including the empty case. If the park were - // written only when roles exist, an entity that was restored and later - // soft-deleted again with no roles would leave the first park as the - // newest one, and the second restore would replay roles the entity no - // longer had. One row per soft-delete keeps "newest park" and "newest - // soft-delete" the same event. - await logDataQuality(env, entityId, "soft_deleted_roles", [], source, actorEmail, roles.results ?? []); - await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entityId).run(); - } - catch (e) { - console.warn("entity_roles delete during soft-delete failed", entityId, e.message); - } - await logDataQuality(env, entityId, "soft_deleted", reasons, source, actorEmail); -} -export async function restoreEntity(env, entityId, actorEmail) { - await env.DB.prepare(`UPDATE u_entities - SET status = 'active', - deleted_reason = NULL, - updated_at = datetime('now') - WHERE id = ?`).bind(entityId).run(); - // Put back the roles the soft delete removed. Restoring status alone - // returned an entity with no roles, so it never reappeared on any - // role-filtered surface — the restore looked like it worked and did not. - let rolesRestored = 0; - try { - const parked = await env.DB.prepare(`SELECT reasons_json FROM data_quality_log - WHERE entity_id = ? AND issue = 'soft_deleted_roles' - ORDER BY detected_at DESC, id DESC LIMIT 1`).bind(entityId).first(); - const rows = parked?.reasons_json ? JSON.parse(parked.reasons_json) : []; - for (const r of Array.isArray(rows) ? rows : []) { - if (!r || typeof r.role !== "string" || !r.role) - continue; - await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) - VALUES (?, ?, ?, ?, ?) - ON CONFLICT(entity_id, role) DO NOTHING`).bind(entityId, r.role, r.is_primary ?? 0, r.source ?? "restore", r.confidence ?? 0.5).run(); - rolesRestored += 1; - } - } - catch (e) { - // A restore that cannot replay the roles is still better than none — - // but it must not look clean, so it is recorded rather than swallowed. - await logDataQuality(env, entityId, "restore_roles_failed", [e.message], "operator", actorEmail); - } - await logDataQuality(env, entityId, "restored", [`roles_restored:${rolesRestored}`], "operator", actorEmail); -} -export async function purgeEntity(env, entityId, actorEmail) { - // Best-effort cascade across the optional referencing tables; each - // wrapped in its own try/catch so a missing table doesn't block the - // primary delete. Per the Task #14 honest-degradation pattern. - const cascades = [ - `DELETE FROM facts WHERE entity_id = ?`, - `DELETE FROM rel_edges WHERE src_entity_id = ? OR dst_entity_id = ?`, - `DELETE FROM channels WHERE entity_id = ?`, - `DELETE FROM entity_roles WHERE entity_id = ?`, - `DELETE FROM entity_history WHERE entity_id = ?`, - `DELETE FROM entity_legacy_map WHERE entity_id = ?`, - ]; - for (const sql of cascades) { - try { - if (sql.includes("OR dst_entity_id")) { - await env.DB.prepare(sql).bind(entityId, entityId).run(); - } - else { - await env.DB.prepare(sql).bind(entityId).run(); - } - } - catch (e) { - // table-missing or FK noise — log and continue - console.warn("purge cascade failed", sql.slice(0, 40), e.message); - } - } - // Log BEFORE the final delete so the audit trail survives even if - // the row-delete races a concurrent reader. data_quality_log keeps - // entity_id as TEXT (no FK), so the row remains queryable. - await logDataQuality(env, entityId, "purged", [], "operator", actorEmail); - await env.DB.prepare(`DELETE FROM u_entities WHERE id = ?`).bind(entityId).run(); -} -// --------------------------------------------------------------------------- -// Task #6: reclassify a person row that is actually an organization. The -// row is FLIPPED in place (kind person→org) so it leaves the People list -// and joins the org world — non-destructive and reversible (the row, its -// facts and relationships are preserved). When a domain/website is known -// we dual-write a `firms` row so it also surfaces in the Firms list; -// upsertFirm's syncFirmToEntity re-resolves the firm to THIS now-org -// entity via the primary_domain match, so no duplicate entity is minted. -// HONEST DEGRADATION: with no domain/website we cannot dedupe a firm row -// (name-only matching mints duplicates), so we skip it and record the gap -// rather than guessing — the entity still leaves People as an org. -// --------------------------------------------------------------------------- -export async function reclassifyPersonAsOrg(env, entity, orgRole, reasons, source, actorEmail) { - // 1. Flip kind in place — removes it from the People list immediately. - await env.DB.prepare(`UPDATE u_entities SET kind = 'org', updated_at = datetime('now') WHERE id = ?`).bind(entity.id).run(); - // 2. Swap person/investor roles for the inferred org role. Capture the - // prior role set into the audit trail FIRST so the reclassification is - // fully reversible: an operator (or a rollback) can restore the original - // roles from the data_quality_log row, not just flip the kind back. - let priorRoles = []; - try { - const existing = await env.DB.prepare(`SELECT role FROM entity_roles WHERE entity_id = ?`).bind(entity.id).all(); - priorRoles = (existing.results ?? []).map((x) => x.role); - await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entity.id).run(); - await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) - VALUES (?, ?, 1, ?, 1) - ON CONFLICT(entity_id, role) DO UPDATE SET is_primary = 1`).bind(entity.id, orgRole, source).run(); - } - catch (e) { - console.warn("reclassify role swap failed", entity.id, e.message); - } - // 3. Dual-write a firms row when we can dedupe by domain/website. - let firmListed = false; - if (entity.display_name && (entity.primary_domain || entity.primary_url)) { - try { - const { upsertFirm } = await import("../scraper/firms_upsert.js"); - await upsertFirm(env, { - name: entity.display_name, - domain: entity.primary_domain ?? null, - website: entity.primary_url ?? null, - kind: orgRoleToFirmKind(orgRole), - }, source); - firmListed = true; - } - catch (e) { - console.warn("reclassify upsertFirm failed", entity.id, e.message); - } - } - await logDataQuality(env, entity.id, "reclassified", [...reasons, `org_role:${orgRole}`, - priorRoles.length ? `prior_roles:${priorRoles.join(",")}` : "prior_roles:none", - firmListed ? "firm_listed" : "firm_row_skipped_no_domain"], source, actorEmail); - return { reclassified: true, firm_listed: firmListed }; -} -export async function runCleanupSweep(env, opts = {}) { - const mode = opts.mode ?? "recent"; - const lookback = opts.lookbackHours ?? 24; - const limit = opts.limit ?? 5000; - const source = opts.source ?? (mode === "all" ? "oneoff_cleanup" : "cron_sweep"); - const where = mode === "all" - ? `status NOT IN ('soft_deleted','merged')` - : `status NOT IN ('soft_deleted','merged') AND created_at >= datetime('now', '-${lookback} hours')`; - const rows = await env.DB.prepare(`SELECT id, kind, display_name, primary_url, primary_domain, primary_email_key, primary_linkedin_key - FROM u_entities - WHERE ${where} - ORDER BY created_at DESC - LIMIT ?`).bind(limit).all(); - const items = rows.results ?? []; - const byReason = {}; - let flagged = 0; - let softDeleted = 0; - let reclassified = 0; - let needsReview = 0; - for (const r of items) { - // Task #6: organization-name disambiguation for `person` rows. - // Orgs scraped as people are RECLASSIFIED into the org world (and the - // Firms list) rather than soft-deleted. A strong personal identifier - // (personal LinkedIn /in/ or an email) contradicting an org-suffix - // name is flagged for operator review instead of auto-flipped — never - // destroy a likely real person. Junk / plausible-person / uncertain - // names fall through to the existing garbage + orphan path below so - // no prior behavior regresses. - if (r.kind === "person") { - const cls = classifyPersonName(r.display_name); - if (cls.verdict === "organization" && cls.orgRole) { - const personalLinkedin = !!r.primary_linkedin_key && /(^|\/)in\//i.test(r.primary_linkedin_key); - const hasEmail = !!r.primary_email_key; - if (personalLinkedin || hasEmail) { - for (const code of cls.reasons) - byReason[`review_${code}`] = (byReason[`review_${code}`] ?? 0) + 1; - await logDataQuality(env, r.id, "needs_review", [...cls.reasons, "org_name_with_person_signal"], source, opts.actorEmail ?? null); - needsReview += 1; - continue; - } - try { - await reclassifyPersonAsOrg(env, { id: r.id, display_name: r.display_name, primary_url: r.primary_url, primary_domain: r.primary_domain }, cls.orgRole, cls.reasons, source, opts.actorEmail ?? null); - reclassified += 1; - byReason[`reclassified_${cls.orgRole}`] = (byReason[`reclassified_${cls.orgRole}`] ?? 0) + 1; - } - catch (e) { - console.warn("sweep reclassify failed", r.id, e.message); - } - continue; - } - } - // Route through evaluateEntity so the AI second opinion fires for - // ambiguous 30–60 char names (when env.AI is bound). Honors - // skipAi for unit tests + operator-requested fast sweeps. - const evald = await evaluateEntity(env, { - kind: r.kind, display_name: r.display_name, - primary_url: r.primary_url, primary_domain: r.primary_domain, - primary_email_key: r.primary_email_key, primary_linkedin_key: r.primary_linkedin_key, - }, { skipAi: opts.skipAi }); - let reasons = evald.reasons; - let flag = evald.is_garbage; - if (!flag) { - // Structural rule — only meaningful when the entity is not - // brand-new (otherwise the crawler may still be writing joins). - const orphan = await isStructurallyOrphan(env, r.id, { minAgeHours: 24 }); - if (orphan.orphan) { - flag = true; - reasons = orphan.reasons; - } - } - if (!flag) - continue; - flagged += 1; - for (const code of reasons) - byReason[code] = (byReason[code] ?? 0) + 1; - try { - await softDeleteEntity(env, r.id, reasons, source, opts.actorEmail ?? null); - softDeleted += 1; - } - catch (e) { - console.warn("sweep soft-delete failed", r.id, e.message); - } - } - const result = { - scanned: items.length, flagged, soft_deleted: softDeleted, - reclassified, needs_review: needsReview, - by_reason: byReason, bounded: items.length >= limit, - }; - console.log("garbage.cleanup_sweep", JSON.stringify({ mode, ...result })); - return result; -} diff --git a/apps/worker/test-dist-q/entities/model.js b/apps/worker/test-dist-q/entities/model.js deleted file mode 100644 index a09d5fab..00000000 --- a/apps/worker/test-dist-q/entities/model.js +++ /dev/null @@ -1,6 +0,0 @@ -// TS types matching the unified entity DDL (migrations 200–208). -// Task #4: re-export the rich-profile write helpers under a single -// EntityService facade so callers can `import { EntityService } from -// "./entities/model"` for every structured profile write. -export { EntityService } from "./profile"; -export { PREDICATE_REGISTRY, PREDICATE_MAP, EMITTED_PREDICATES, getPredicateMeta, } from "./profile-predicates"; diff --git a/apps/worker/test-dist-q/entities/normalize.js b/apps/worker/test-dist-q/entities/normalize.js deleted file mode 100644 index 84318cba..00000000 --- a/apps/worker/test-dist-q/entities/normalize.js +++ /dev/null @@ -1,120 +0,0 @@ -// Canonical key generators for the unified entity model. Channels are -// keyed by (kind, canonical) and equality must collapse trivial -// presentation differences — case, whitespace, '+suffix' in emails, -// formatting characters in phone numbers, query strings on LinkedIn URLs. -export function canonicalEmail(raw) { - if (!raw) - return null; - const s = String(raw).trim().toLowerCase(); - const m = /^([^@\s]+)@([a-z0-9.-]+\.[a-z]{2,})$/i.exec(s); - if (!m) - return null; - // Strip '+suffix' from the local part (Gmail-style). - const local = m[1].split("+")[0]; - return `${local}@${m[2]}`; -} -export function canonicalPhone(raw) { - if (!raw) - return null; - const digits = String(raw).replace(/[^\d+]/g, ""); - if (!digits) - return null; - // Best-effort E.164: keep leading + if present, else assume already - // includes country code. - return digits.startsWith("+") ? digits : `+${digits.replace(/^\++/, "")}`; -} -export function canonicalLinkedin(raw) { - if (!raw) - return null; - let s = String(raw).trim(); - if (!s) - return null; - // Accept handle-only ("janedoe") or full URL. - if (!/^https?:\/\//i.test(s) && !s.includes("/")) { - return `/in/${s.toLowerCase()}`; - } - try { - const u = new URL(s.startsWith("http") ? s : `https://${s}`); - if (!/linkedin\.com$/i.test(u.hostname) && !/(^|\.)linkedin\.com$/i.test(u.hostname)) - return null; - const p = u.pathname.replace(/\/+$/, "").toLowerCase(); - // /in/ | /company/ | /school/ - const m = /^\/(in|company|school|pub)\/([^/?#]+)/.exec(p); - if (!m) - return null; - return `/${m[1] === "pub" ? "in" : m[1]}/${m[2]}`; - } - catch { - return null; - } -} -export function canonicalTwitter(raw) { - if (!raw) - return null; - const s = String(raw).trim().replace(/^@/, ""); - if (!s) - return null; - if (/^https?:\/\//i.test(s)) { - try { - const u = new URL(s); - const m = /^\/([A-Za-z0-9_]{1,15})\/?$/.exec(u.pathname); - return m ? m[1].toLowerCase() : null; - } - catch { - return null; - } - } - return /^[A-Za-z0-9_]{1,15}$/.test(s) ? s.toLowerCase() : null; -} -export function canonicalGithub(raw) { - if (!raw) - return null; - const s = String(raw).trim().replace(/^@/, ""); - if (!s) - return null; - if (/^https?:\/\//i.test(s)) { - try { - const u = new URL(s); - const m = /^\/([A-Za-z0-9-]{1,39})\/?$/.exec(u.pathname); - return m ? m[1].toLowerCase() : null; - } - catch { - return null; - } - } - return /^[A-Za-z0-9-]{1,39}$/.test(s) ? s.toLowerCase() : null; -} -export function canonicalDomain(raw) { - if (!raw) - return null; - let s = String(raw).trim().toLowerCase(); - if (!s) - return null; - if (/^https?:\/\//.test(s)) { - try { - s = new URL(s).hostname; - } - catch { - return null; - } - } - s = s.replace(/^www\./, "").replace(/\/.*$/, ""); - return /^[a-z0-9.-]+\.[a-z]{2,}$/.test(s) ? s : null; -} -export function canonicalUrl(raw) { - if (!raw) - return null; - try { - const u = new URL(String(raw).trim()); - u.hash = ""; - return u.toString(); - } - catch { - return null; - } -} -export async function sha256(s) { - const buf = new TextEncoder().encode(s); - const digest = await crypto.subtle.digest("SHA-256", buf); - return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); -} diff --git a/apps/worker/test-dist-q/entities/profile-predicates.js b/apps/worker/test-dist-q/entities/profile-predicates.js deleted file mode 100644 index f88842dc..00000000 --- a/apps/worker/test-dist-q/entities/profile-predicates.js +++ /dev/null @@ -1,218 +0,0 @@ -// Task #4: Predicate registry — single source of truth. -// -// The same list is mirrored into `predicate_registry` by migration 328. -// Every predicate string that `profile.ts` ever passes to `insertFact` -// MUST appear in this array; the CI smoke test (test/profile.test.mjs) -// enforces it. Adding a new predicate is a TWO-file change: -// 1. append it here AND -// 2. add the matching `INSERT OR IGNORE INTO predicate_registry` row in -// migration 328_predicate_registry.sql. -// -// `value_type` is informational metadata for the UI formatter — it does -// not constrain the runtime shape of the fact value_json (those shapes -// live in profile-shapes.ts). -export const PREDICATE_REGISTRY = [ - // ---- identity (rich-profile) ------------------------------------------- - { predicate: "person.identity", label: "Identity snapshot", icon: "user", formatter: "json", category: "identity", value_type: "json", description: "Snapshot of person_identity row." }, - { predicate: "person.identity.full_name", label: "Full name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Legal or commonly used full name." }, - { predicate: "person.identity.preferred_name", label: "Preferred name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "What the person prefers to be called." }, - { predicate: "person.identity.pronouns", label: "Pronouns", icon: "user-circle", formatter: "text", category: "identity", value_type: "json", description: "Pronouns triple (subject/object/possessive)." }, - { predicate: "person.identity.birth_year", label: "Birth year", icon: "cake", formatter: "year", category: "identity", value_type: "year", description: "Year of birth (no full DOB stored)." }, - { predicate: "person.identity.nationality", label: "Nationality", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO-3166-1 alpha-2 nationality." }, - { predicate: "person.identity.languages", label: "Languages", icon: "languages", formatter: "list", category: "identity", value_type: "json", description: "Spoken languages with proficiency." }, - { predicate: "person.identity.timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "IANA tz database name." }, - { predicate: "person.identity.location_city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Current city of residence." }, - { predicate: "person.identity.location_country", label: "Country", icon: "globe", formatter: "flag", category: "identity", value_type: "country_iso2", description: "Current country of residence." }, - { predicate: "person.identity.headshot_url", label: "Headshot", icon: "image", formatter: "avatar", category: "identity", value_type: "url", description: "Public headshot image URL." }, - // ---- career ------------------------------------------------------------- - { predicate: "person.career", label: "Career entry", icon: "briefcase", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a career_history row." }, - // ---- board -------------------------------------------------------------- - { predicate: "person.board_seat", label: "Board seat", icon: "users", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a board_seats row." }, - // ---- education ---------------------------------------------------------- - { predicate: "person.education", label: "Education", icon: "graduation-cap", formatter: "json", category: "education", value_type: "json", description: "Snapshot of an education_history row." }, - // ---- family ------------------------------------------------------------- - { predicate: "person.family_tie", label: "Family tie", icon: "heart", formatter: "json", category: "family", value_type: "json", description: "Snapshot of a family_ties row." }, - // ---- conference --------------------------------------------------------- - { predicate: "person.conference", label: "Conference", icon: "calendar", formatter: "json", category: "conference", value_type: "json", description: "Snapshot of a conference_attendance row." }, - // ---- preferences (one predicate per documented preference_key) --------- - { predicate: "person.preference.communication_channel", label: "Communication channel", icon: "message-circle", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact channel (email/text/dm/voice)." }, - { predicate: "person.preference.contact_time", label: "Best time to reach", icon: "clock", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact window or day." }, - { predicate: "person.preference.meeting_format", label: "Meeting format", icon: "video", formatter: "text", category: "preference", value_type: "text", description: "In-person, video, phone, async." }, - { predicate: "person.preference.gift_dietary", label: "Dietary", icon: "salad", formatter: "text", category: "preference", value_type: "text", description: "Vegan/vegetarian/keto/halal/etc." }, - { predicate: "person.preference.gift_allergies", label: "Allergies", icon: "alert-triangle", formatter: "list", category: "preference", value_type: "json", description: "Food/material allergies to avoid for gifting." }, - { predicate: "person.preference.coffee_order", label: "Coffee order", icon: "coffee", formatter: "text", category: "preference", value_type: "text", description: "Usual coffee order." }, - { predicate: "person.preference.travel_class", label: "Travel class", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Preferred flight cabin class." }, - { predicate: "person.preference.hotel_brand", label: "Hotel brand", icon: "bed", formatter: "text", category: "preference", value_type: "text", description: "Preferred hotel chain or brand." }, - { predicate: "person.preference.airline_status", label: "Airline status", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Frequent-flyer status / preferred carrier." }, - // ---- interests (one per category) -------------------------------------- - { predicate: "person.interest.topic", label: "Topic", icon: "tag", formatter: "badge", category: "interest", value_type: "text", description: "Topic of interest." }, - { predicate: "person.interest.sport", label: "Sport", icon: "trophy", formatter: "badge", category: "interest", value_type: "text", description: "Sport played or followed." }, - { predicate: "person.interest.team", label: "Team", icon: "shield", formatter: "badge", category: "interest", value_type: "text", description: "Favorite team." }, - { predicate: "person.interest.book", label: "Book", icon: "book", formatter: "text", category: "interest", value_type: "text", description: "Book the person recommends or read." }, - { predicate: "person.interest.author", label: "Author", icon: "feather", formatter: "text", category: "interest", value_type: "text", description: "Favorite author." }, - { predicate: "person.interest.podcast", label: "Podcast", icon: "mic", formatter: "text", category: "interest", value_type: "text", description: "Podcast the person listens to." }, - { predicate: "person.interest.music", label: "Music genre", icon: "music", formatter: "badge", category: "interest", value_type: "text", description: "Preferred music genre." }, - { predicate: "person.interest.artist", label: "Artist", icon: "music", formatter: "text", category: "interest", value_type: "text", description: "Favorite musical artist." }, - { predicate: "person.interest.film", label: "Film", icon: "film", formatter: "text", category: "interest", value_type: "text", description: "Favorite film." }, - { predicate: "person.interest.show", label: "TV show", icon: "tv", formatter: "text", category: "interest", value_type: "text", description: "Favorite TV show." }, - { predicate: "person.interest.hobby", label: "Hobby", icon: "puzzle", formatter: "badge", category: "interest", value_type: "text", description: "Hobby outside of work." }, - { predicate: "person.interest.cause", label: "Cause", icon: "heart-handshake", formatter: "badge", category: "interest", value_type: "text", description: "Cause the person supports." }, - // ---- lifestyle signals ------------------------------------------------- - { predicate: "person.lifestyle.runs", label: "Runs", icon: "footprints", formatter: "text", category: "lifestyle", value_type: "json", description: "Is a runner." }, - { predicate: "person.lifestyle.cycles", label: "Cycles", icon: "bike", formatter: "text", category: "lifestyle", value_type: "json", description: "Cycles." }, - { predicate: "person.lifestyle.surfs", label: "Surfs", icon: "waves", formatter: "text", category: "lifestyle", value_type: "json", description: "Surfs." }, - { predicate: "person.lifestyle.skis", label: "Skis", icon: "snowflake", formatter: "text", category: "lifestyle", value_type: "json", description: "Skis or snowboards." }, - { predicate: "person.lifestyle.golfs", label: "Golfs", icon: "flag", formatter: "text", category: "lifestyle", value_type: "json", description: "Plays golf." }, - { predicate: "person.lifestyle.yoga", label: "Yoga", icon: "activity", formatter: "text", category: "lifestyle", value_type: "json", description: "Practices yoga." }, - { predicate: "person.lifestyle.meditates", label: "Meditates", icon: "leaf", formatter: "text", category: "lifestyle", value_type: "json", description: "Meditates regularly." }, - { predicate: "person.lifestyle.cooks", label: "Cooks", icon: "chef-hat", formatter: "text", category: "lifestyle", value_type: "json", description: "Cooks for fun." }, - { predicate: "person.lifestyle.collects", label: "Collects", icon: "package", formatter: "text", category: "lifestyle", value_type: "json", description: "Collects something (art, wine, watches…)." }, - { predicate: "person.lifestyle.pet", label: "Pet", icon: "paw-print", formatter: "text", category: "lifestyle", value_type: "json", description: "Has a pet." }, - { predicate: "person.lifestyle.marathon", label: "Marathon", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed marathon." }, - { predicate: "person.lifestyle.ironman", label: "Ironman", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed Ironman triathlon." }, - // ---- travel ------------------------------------------------------------- - { predicate: "person.travel.frequent_city", label: "Frequent city", icon: "map-pin", formatter: "text", category: "travel", value_type: "text", description: "City the person frequently travels to." }, - { predicate: "person.travel.home_base", label: "Home base", icon: "home", formatter: "text", category: "travel", value_type: "text", description: "Stated home base." }, - { predicate: "person.travel.recent_trip", label: "Recent trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Recent trip place + date window." }, - { predicate: "person.travel.upcoming_trip", label: "Upcoming trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Announced upcoming trip." }, - { predicate: "person.travel.airport_hub", label: "Airport hub", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Home airport (IATA code)." }, - // ---- goals -------------------------------------------------------------- - { predicate: "person.goal.short_term", label: "Short-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for <12 months." }, - { predicate: "person.goal.long_term", label: "Long-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for 12+ months." }, - { predicate: "person.goal.hiring", label: "Hiring goal", icon: "user-plus", formatter: "text", category: "goal", value_type: "text", description: "Stated hiring need." }, - { predicate: "person.goal.fundraising", label: "Fundraising goal", icon: "dollar-sign", formatter: "text", category: "goal", value_type: "text", description: "Stated fundraising plan." }, - { predicate: "person.goal.investing_thesis", label: "Investing thesis", icon: "lightbulb", formatter: "text", category: "goal", value_type: "text", description: "Stated investing thesis." }, - { predicate: "person.goal.expansion_market", label: "Expansion market", icon: "map", formatter: "text", category: "goal", value_type: "text", description: "Stated market expansion." }, - // ---- conversation hooks ------------------------------------------------ - { predicate: "person.hook.recent_news", label: "Recent news", icon: "newspaper", formatter: "text", category: "hook", value_type: "text", description: "Recent news about the person." }, - { predicate: "person.hook.shared_connection", label: "Shared connection", icon: "users", formatter: "text", category: "hook", value_type: "text", description: "Mutual contact you can mention." }, - { predicate: "person.hook.shared_school", label: "Shared school", icon: "graduation-cap", formatter: "text", category: "hook", value_type: "text", description: "Shared alma mater." }, - { predicate: "person.hook.shared_employer", label: "Shared employer", icon: "briefcase", formatter: "text", category: "hook", value_type: "text", description: "Shared former or current employer." }, - { predicate: "person.hook.shared_interest", label: "Shared interest", icon: "tag", formatter: "text", category: "hook", value_type: "text", description: "Shared interest or hobby." }, - { predicate: "person.hook.recent_post", label: "Recent post", icon: "message-square", formatter: "text", category: "hook", value_type: "text", description: "Recent public post by the person." }, - { predicate: "person.hook.life_event", label: "Life event", icon: "sparkles", formatter: "text", category: "hook", value_type: "text", description: "Life event (job change, baby, move)." }, - { predicate: "person.hook.opinion_quoted", label: "Opinion quoted", icon: "quote", formatter: "text", category: "hook", value_type: "text", description: "Public statement on a topic." }, - // ---- appreciation ------------------------------------------------------ - { predicate: "person.appreciation.compliment_topic", label: "Compliment topic", icon: "thumbs-up", formatter: "text", category: "appreciation", value_type: "text", description: "Topic the person likes to be complimented on." }, - { predicate: "person.appreciation.gift_idea", label: "Gift idea", icon: "gift", formatter: "text", category: "appreciation", value_type: "text", description: "Concrete gift idea." }, - { predicate: "person.appreciation.charity_supported", label: "Charity supported", icon: "heart", formatter: "text", category: "appreciation", value_type: "text", description: "Charity the person supports." }, - { predicate: "person.appreciation.cause_advocated", label: "Cause advocated", icon: "megaphone", formatter: "text", category: "appreciation", value_type: "text", description: "Cause the person publicly advocates." }, - { predicate: "person.appreciation.recognition_received", label: "Recognition received", icon: "award", formatter: "text", category: "appreciation", value_type: "text", description: "Public award/recognition received." }, - // ---- legacy / extractor predicates (so the UI never renders a raw key) - - { predicate: "name", label: "Name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Display name (extractor field)." }, - { predicate: "title", label: "Title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Job title (extractor field)." }, - { predicate: "role", label: "Role", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Functional role." }, - { predicate: "employer", label: "Employer", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Current employer name." }, - { predicate: "company", label: "Company", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Alias for employer in some extractors." }, - { predicate: "headline", label: "Headline", icon: "type", formatter: "text", category: "identity", value_type: "text", description: "LinkedIn-style headline." }, - { predicate: "summary", label: "Summary", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Bio / summary paragraph." }, - { predicate: "bio", label: "Bio", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Profile bio." }, - { predicate: "description", label: "Description", icon: "file-text", formatter: "text", category: "firm", value_type: "text", description: "Org/company description." }, - { predicate: "location", label: "Location", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Stated location string." }, - { predicate: "city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "City (extractor field)." }, - { predicate: "region", label: "Region", icon: "map", formatter: "text", category: "identity", value_type: "text", description: "State/region." }, - { predicate: "country", label: "Country", icon: "globe", formatter: "text", category: "identity", value_type: "text", description: "Country (name string)." }, - { predicate: "country_iso2", label: "Country (ISO)", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO 3166-1 alpha-2 country." }, - { predicate: "timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "Timezone (extractor field)." }, - { predicate: "email", label: "Email", icon: "mail", formatter: "link", category: "contact", value_type: "email", description: "Email address." }, - { predicate: "phone", label: "Phone", icon: "phone", formatter: "text", category: "contact", value_type: "text", description: "Phone number (E.164 preferred)." }, - { predicate: "linkedin_url", label: "LinkedIn", icon: "linkedin", formatter: "link", category: "contact", value_type: "url", description: "LinkedIn profile URL." }, - { predicate: "twitter_url", label: "Twitter", icon: "twitter", formatter: "link", category: "contact", value_type: "url", description: "Twitter/X profile URL." }, - { predicate: "twitter_handle", label: "Twitter handle", icon: "twitter", formatter: "text", category: "contact", value_type: "text", description: "Twitter/X handle." }, - { predicate: "github_url", label: "GitHub", icon: "github", formatter: "link", category: "contact", value_type: "url", description: "GitHub profile URL." }, - { predicate: "github_handle", label: "GitHub handle", icon: "github", formatter: "text", category: "contact", value_type: "text", description: "GitHub handle." }, - { predicate: "website", label: "Website", icon: "link", formatter: "link", category: "contact", value_type: "url", description: "Personal/firm website URL." }, - { predicate: "primary_domain", label: "Primary domain", icon: "globe", formatter: "text", category: "firm", value_type: "text", description: "Canonical apex domain." }, - { predicate: "sector", label: "Sector", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Sector / vertical tag." }, - { predicate: "stage", label: "Stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Investment stage focus." }, - { predicate: "check_size_min_usd", label: "Min check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Minimum typical check size." }, - { predicate: "check_size_max_usd", label: "Max check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Maximum typical check size." }, - { predicate: "fund_size_usd", label: "Fund size (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent fund size." }, - { predicate: "founded_year", label: "Founded year", icon: "calendar", formatter: "year", category: "firm", value_type: "year", description: "Year founded." }, - { predicate: "founded_at", label: "Founded", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Founding date (full)." }, - { predicate: "funding_stage", label: "Funding stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Latest funding stage." }, - { predicate: "total_funding_usd", label: "Total funding (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Total funding raised." }, - { predicate: "last_round_usd", label: "Last round (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent round amount." }, - { predicate: "hq_city", label: "HQ city", icon: "map-pin", formatter: "text", category: "firm", value_type: "text", description: "HQ city." }, - { predicate: "hq_country_iso2", label: "HQ country (ISO)", icon: "flag", formatter: "flag", category: "firm", value_type: "country_iso2", description: "HQ country ISO code." }, - { predicate: "employees", label: "Employees", icon: "users", formatter: "text", category: "firm", value_type: "number", description: "Employee headcount or band." }, - { predicate: "industry", label: "Industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry classification." }, - { predicate: "display_name", label: "Display name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Canonical display name." }, - // ---- Task #1: SEC EDGAR deep-adapter predicates ----------------------- - // Mirror in migration 349_sec_edgar.sql — two-file change enforced by - // test/profile.test.mjs. - { predicate: "sec.cik", label: "SEC CIK", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC Central Index Key (10-digit, zero-padded)." }, - { predicate: "sec.crd", label: "SEC CRD", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Investment Adviser CRD# from Form ADV." }, - { predicate: "sec.cusip", label: "CUSIP", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Committee on Uniform Securities Identification Procedures code." }, - { predicate: "sec.ticker", label: "Ticker", icon: "trending-up", formatter: "badge", category: "identity", value_type: "text", description: "Public ticker symbol." }, - { predicate: "sec.sec_file_number", label: "SEC file number", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC file number (e.g. 801-12345 for advisers)." }, - { predicate: "sec.fund_id_807", label: "SEC fund ID", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC fund identifier (807-XXXXXXXX)." }, - { predicate: "aum_usd", label: "AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Assets under management in USD." }, - { predicate: "sec.form_adv.filed_at", label: "Last Form ADV filed", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Most recent Form ADV acceptance date." }, - { predicate: "sec.form_adv.fund", label: "Form ADV fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Fund disclosed on Schedule D §7.B.(1)." }, - { predicate: "sec.form_d.round", label: "Form D round", icon: "dollar-sign", formatter: "json", category: "firm", value_type: "json", description: "Private placement disclosed on Form D." }, - { predicate: "sec.form_d.issuer_industry", label: "Form D industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry group declared on Form D." }, - { predicate: "sec.13f.holding", label: "13F holding", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Equity position disclosed on Form 13F-HR." }, - { predicate: "sec.13f.filer_aum_usd", label: "13F filer AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Aggregate USD value of 13F holdings (proxy AUM)." }, - { predicate: "sec.13d.beneficial_owner", label: "13D beneficial owner", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Schedule 13D 5%+ beneficial-ownership disclosure." }, - { predicate: "sec.form4.insider_trade", label: "Insider trade", icon: "arrow-up-down", formatter: "json", category: "firm", value_type: "json", description: "Form 4 §16 insider transaction." }, - { predicate: "sec.form4.officer_title", label: "Officer title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Officer title declared on Form 4 (when reporter is officer)." }, - { predicate: "sec.s1.ipo_intent", label: "S-1 IPO intent", icon: "rocket", formatter: "text", category: "firm", value_type: "text", description: "Company filed Form S-1 (IPO registration)." }, - { predicate: "sec.s1.underwriter", label: "IPO underwriter", icon: "briefcase", formatter: "text", category: "firm", value_type: "text", description: "Underwriter listed on Form S-1." }, - { predicate: "sec.8k.material_event", label: "8-K material event", icon: "alert-circle", formatter: "json", category: "firm", value_type: "json", description: "Form 8-K current report item." }, - { predicate: "sec.10k.revenue_usd", label: "10-K revenue (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Annual revenue from Form 10-K." }, - { predicate: "sec.10k.net_income_usd", label: "10-K net income", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Net income from Form 10-K." }, - { predicate: "sec.10k.fiscal_year_end", label: "10-K fiscal year-end", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Fiscal year-end from Form 10-K." }, - { predicate: "sec.10k.executive", label: "10-K executive", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Named executive officer compensation from Form 10-K." }, - { predicate: "sec.pf.fund", label: "Form PF fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Private fund disclosure from Form PF (large private fund adviser)." }, - { predicate: "sec.gp_disclosed", label: "GP disclosed (SEC)", icon: "user-check", formatter: "text", category: "firm", value_type: "text", description: "GP / control-person disclosed on a SEC filing." }, -]; -export const PREDICATE_MAP = Object.freeze(Object.fromEntries(PREDICATE_REGISTRY.map((p) => [p.predicate, p]))); -export function getPredicateMeta(predicate) { - return PREDICATE_MAP[predicate] ?? null; -} -// Helpers in `profile.ts` MUST only emit predicates listed here. -// The smoke test asserts EMITTED_PREDICATES ⊆ PREDICATE_REGISTRY keys. -export const EMITTED_PREDICATES = Object.freeze([ - // static (one per helper) - "person.identity", - "person.career", - "person.board_seat", - "person.education", - "person.family_tie", - "person.conference", - // dynamic: person.preference.{key} - "person.preference.communication_channel", - "person.preference.contact_time", - "person.preference.meeting_format", - "person.preference.gift_dietary", - "person.preference.gift_allergies", - "person.preference.coffee_order", - "person.preference.travel_class", - "person.preference.hotel_brand", - "person.preference.airline_status", - // dynamic: person.interest.{category} - "person.interest.topic", "person.interest.sport", "person.interest.team", - "person.interest.book", "person.interest.author", "person.interest.podcast", - "person.interest.music", "person.interest.artist", "person.interest.film", - "person.interest.show", "person.interest.hobby", "person.interest.cause", - // dynamic: person.lifestyle.{key} - "person.lifestyle.runs", "person.lifestyle.cycles", "person.lifestyle.surfs", - "person.lifestyle.skis", "person.lifestyle.golfs", "person.lifestyle.yoga", - "person.lifestyle.meditates", "person.lifestyle.cooks", "person.lifestyle.collects", - "person.lifestyle.pet", "person.lifestyle.marathon", "person.lifestyle.ironman", - // dynamic: person.travel.{kind} - "person.travel.frequent_city", "person.travel.home_base", - "person.travel.recent_trip", "person.travel.upcoming_trip", "person.travel.airport_hub", - // dynamic: person.goal.{kind} - "person.goal.short_term", "person.goal.long_term", "person.goal.hiring", - "person.goal.fundraising", "person.goal.investing_thesis", "person.goal.expansion_market", - // dynamic: person.hook.{kind} - "person.hook.recent_news", "person.hook.shared_connection", "person.hook.shared_school", - "person.hook.shared_employer", "person.hook.shared_interest", "person.hook.recent_post", - "person.hook.life_event", "person.hook.opinion_quoted", - // dynamic: person.appreciation.{kind} - "person.appreciation.compliment_topic", "person.appreciation.gift_idea", - "person.appreciation.charity_supported", "person.appreciation.cause_advocated", - "person.appreciation.recognition_received", -]); diff --git a/apps/worker/test-dist-q/entities/profile-shapes.js b/apps/worker/test-dist-q/entities/profile-shapes.js deleted file mode 100644 index ab1d4f7f..00000000 --- a/apps/worker/test-dist-q/entities/profile-shapes.js +++ /dev/null @@ -1,4 +0,0 @@ -// Task #4: Typed shapes for the JSON columns on rich-profile tables. -// Every helper in `profile.ts` serializes through these shapes; nothing -// passes raw `unknown` through `JSON.stringify`. -export {}; diff --git a/apps/worker/test-dist-q/entities/profile.js b/apps/worker/test-dist-q/entities/profile.js deleted file mode 100644 index 7de9b64d..00000000 --- a/apps/worker/test-dist-q/entities/profile.js +++ /dev/null @@ -1,681 +0,0 @@ -// Task #4: EntityService write helpers for the rich person profile. -// -// Every helper: -// 1. Validates input against the typed shape (profile-shapes.ts). -// 2. Acquires the per-entity DO lock (EntityLock /acquire) so concurrent -// scrapers / OSINT / agent calls don't race on the same entity_id. -// 3. Upserts the structured row using a stable natural key -// (ON CONFLICT(...) DO UPDATE). -// 4. Mirrors a canonical row into `facts` via insertFact — the fact's -// hash dedupe key is sha256(entity|predicate|value|source_url) so the -// second call with identical content updates `observed_at` rather -// than creating a duplicate row. -// -// Public-signal constraint: every helper except `setPersonIdentity` (which -// allows operator-asserted rows with isOperatorAsserted=true) refuses to -// write without a source_url. -import { sha256 } from "./normalize"; -import { enqueueSummaryRebuild } from "./summaryQueue"; -import { EMITTED_PREDICATES, PREDICATE_MAP, } from "./profile-predicates"; -// ---- Internal: per-entity lock (token mutex via EntityLock DO) ----------- -// -// Mirrors the OSINT resolver's acquire/release pattern so all rich-profile -// writes for a given entity_id serialize against OSINT, dual-write, and -// any other caller that already uses the same DO id-namespace. -// -// Strict serialization (task contract): when the ENTITY_LOCK binding is -// present we MUST hold the lock for the duration of the write. If the -// DO is unreachable or the lock is currently held by another writer we -// retry with exponential backoff and then fail loudly rather than -// silently racing. The only bypass is when the binding itself is absent -// — that's how the in-memory unit tests run, and it's explicit. -export class ProfileLockError extends Error { - constructor(entityId, reason) { - super(`profile lock for ${entityId}: ${reason}`); - this.name = "ProfileLockError"; - } -} -async function withProfileLock(env, entityId, fn) { - if (!env.ENTITY_LOCK) - return await fn(); - const stub = env.ENTITY_LOCK.get(env.ENTITY_LOCK.idFromName(entityId)); - const token = crypto.randomUUID(); - let acquired = false; - let lastErr = "unknown"; - // 5 attempts, 50 / 100 / 200 / 400 / 800ms backoff = ≤1.55s total. - // Beyond that we surface the error so the caller can decide (retry the - // job, surface to the operator, etc.) rather than corrupt with a - // racy write. - for (let attempt = 0; attempt < 5 && !acquired; attempt++) { - if (attempt > 0) { - await new Promise((r) => setTimeout(r, 50 * 2 ** (attempt - 1))); - } - try { - const res = await stub.fetch("https://lock/acquire", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ token, ttlMs: 60_000 }), - }); - if (res.ok) { - acquired = true; - break; - } - lastErr = `acquire returned ${res.status}`; - } - catch (e) { - lastErr = e.message || "fetch failed"; - } - } - if (!acquired) { - throw new ProfileLockError(entityId, lastErr); - } - try { - return await fn(); - } - finally { - try { - await stub.fetch("https://lock/release", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ token }), - }); - } - catch { /* lock will TTL out after 60s */ } - } -} -// ---- Internal: mirror a structured row into `facts` --------------------- -// -// Centralized projection so every helper produces consistent fact rows. -// -// Dedupe contract (task spec): natural key is (entity_id, predicate, -// source_url) — independent of value. Re-observing the same predicate -// from the same source_url MUST upsert the existing fact (new value, -// refreshed observed_at) rather than create a second row. The hash we -// compute here covers ONLY those three fields, and UNIQUE(hash) on -// `facts` (migration 201) turns a collision into our UPDATE path. -// -// We bypass `insertFact` for the mirror path because its hash includes -// the value (which is correct for raw scraper writes but wrong here — -// it would let value drift create duplicate (entity, predicate, source) -// rows). The summary-rebuild enqueue is still triggered. -// -// Asserts the predicate exists in the registry — catches typos at -// runtime and is defense-in-depth backup to the smoke-test enforcement -// of EMITTED_PREDICATES ⊆ registry. -async function mirrorFact(env, args) { - if (!PREDICATE_MAP[args.predicate]) { - throw new Error(`profile.mirrorFact: predicate "${args.predicate}" is not in PREDICATE_REGISTRY`); - } - const hash = await sha256(`${args.entityId}|${args.predicate}|${args.sourceUrl}`); - const valueText = args.valueText ?? null; - const valueNumber = args.valueNumber ?? null; - const valueJsonStr = args.valueJson != null ? JSON.stringify(args.valueJson) : null; - const confidence = args.confidence ?? 1.0; - const now = args.observedAt ?? new Date().toISOString(); - try { - await env.DB.prepare(`INSERT INTO facts ( - id, entity_id, predicate, value_text, value_number, value_json, - value_entity_id, source_kind, source, evidence_url, confidence, - observed_at, valid_from, valid_to, is_current, hash - ) VALUES (?, ?, ?, ?, ?, ?, NULL, 'enrichment', ?, ?, ?, ?, NULL, NULL, 1, ?)`).bind(crypto.randomUUID(), args.entityId, args.predicate, valueText, valueNumber, valueJsonStr, args.sourceUrl, args.sourceUrl, confidence, now, hash).run(); - } - catch (e) { - const msg = e.message || ""; - if (/UNIQUE/i.test(msg)) { - await env.DB.prepare(`UPDATE facts - SET value_text = ?, value_number = ?, value_json = ?, - confidence = MAX(confidence, ?), - observed_at = ?, is_current = 1 - WHERE hash = ?`).bind(valueText, valueNumber, valueJsonStr, confidence, now, hash).run(); - } - else { - throw e; - } - } - try { - await enqueueSummaryRebuild(env, args.entityId); - } - catch { /* best-effort */ } -} -function requireSourceUrl(helper, url) { - if (!url || typeof url !== "string" || url.trim().length === 0) { - throw new Error(`profile.${helper}: source_url is required (public-signal-only constraint)`); - } - return url; -} -function requireNonEmpty(helper, field, v) { - if (!v || typeof v !== "string" || v.trim().length === 0) { - throw new Error(`profile.${helper}: ${field} is required`); - } - return v.trim(); -} -function nowIso() { return new Date().toISOString(); } -// ========================================================================= -// 1. setPersonIdentity — upsert on entity_id. -// ========================================================================= -export async function setPersonIdentity(env, input) { - requireNonEmpty("setPersonIdentity", "entityId", input.entityId); - const isOperator = input.isOperatorAsserted === true; - if (!isOperator) - requireSourceUrl("setPersonIdentity", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO person_identity ( - entity_id, full_name, preferred_name, pronouns_json, birth_year, - nationality, languages_json, timezone, location_city, location_country, - headshot_url, source_url, is_operator_asserted, confidence, - observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id) DO UPDATE SET - full_name = COALESCE(excluded.full_name, person_identity.full_name), - preferred_name = COALESCE(excluded.preferred_name, person_identity.preferred_name), - pronouns_json = COALESCE(excluded.pronouns_json, person_identity.pronouns_json), - birth_year = COALESCE(excluded.birth_year, person_identity.birth_year), - nationality = COALESCE(excluded.nationality, person_identity.nationality), - languages_json = COALESCE(excluded.languages_json, person_identity.languages_json), - timezone = COALESCE(excluded.timezone, person_identity.timezone), - location_city = COALESCE(excluded.location_city, person_identity.location_city), - location_country = COALESCE(excluded.location_country, person_identity.location_country), - headshot_url = COALESCE(excluded.headshot_url, person_identity.headshot_url), - source_url = COALESCE(excluded.source_url, person_identity.source_url), - is_operator_asserted = MAX(excluded.is_operator_asserted, person_identity.is_operator_asserted), - confidence = MAX(excluded.confidence, person_identity.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(input.entityId, input.fullName ?? null, input.preferredName ?? null, input.pronouns ? JSON.stringify(input.pronouns) : null, input.birthYear ?? null, input.nationality ?? null, input.languages ? JSON.stringify(input.languages) : null, input.timezone ?? null, input.locationCity ?? null, input.locationCountry ?? null, input.headshotUrl ?? null, input.sourceUrl ?? null, isOperator ? 1 : 0, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.identity", - sourceUrl: input.sourceUrl ?? "operator://asserted", - valueJson: { - full_name: input.fullName ?? null, - preferred_name: input.preferredName ?? null, - pronouns: input.pronouns ?? null, - birth_year: input.birthYear ?? null, - nationality: input.nationality ?? null, - languages: input.languages ?? null, - timezone: input.timezone ?? null, - location_city: input.locationCity ?? null, - location_country: input.locationCountry ?? null, - headshot_url: input.headshotUrl ?? null, - is_operator_asserted: isOperator, - }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 2. addCareerEntry — natural key (entity_id, organization_*, started_at). -// ========================================================================= -export async function addCareerEntry(env, input) { - requireNonEmpty("addCareerEntry", "entityId", input.entityId); - requireNonEmpty("addCareerEntry", "organizationName", input.organizationName); - const sourceUrl = requireSourceUrl("addCareerEntry", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO career_history ( - id, entity_id, organization_entity_id, organization_name, role_title, - seniority, department, started_at, ended_at, is_current, summary, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, COALESCE(organization_entity_id,''), COALESCE(started_at,'')) DO UPDATE SET - organization_name = COALESCE(excluded.organization_name, career_history.organization_name), - role_title = COALESCE(excluded.role_title, career_history.role_title), - seniority = COALESCE(excluded.seniority, career_history.seniority), - department = COALESCE(excluded.department, career_history.department), - ended_at = COALESCE(excluded.ended_at, career_history.ended_at), - is_current = excluded.is_current, - summary = COALESCE(excluded.summary, career_history.summary), - confidence = MAX(excluded.confidence, career_history.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.roleTitle ?? null, input.seniority ?? null, input.department ?? null, input.startedAt ?? null, input.endedAt ?? null, input.isCurrent ? 1 : 0, input.summary ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.career", - sourceUrl, - valueJson: { - organization_entity_id: input.organizationEntityId ?? null, - organization_name: input.organizationName, - role_title: input.roleTitle ?? null, - seniority: input.seniority ?? null, - department: input.department ?? null, - started_at: input.startedAt ?? null, - ended_at: input.endedAt ?? null, - is_current: input.isCurrent === true, - }, - confidence: input.confidence, - observedAt: now, - }); - }); - // Task #8: career changes materially affect persona match scores - // (title, seniority, function, employer industry/size/stage). Fire - // the debounced re-match trigger. - try { - const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); - void triggerEntityMatchRefresh(env, input.entityId).catch(() => undefined); - } - catch { /* trigger is best-effort */ } -} -// ========================================================================= -// 3. addBoardSeat — natural key (entity_id, organization_name, started_at). -// ========================================================================= -export async function addBoardSeat(env, input) { - requireNonEmpty("addBoardSeat", "entityId", input.entityId); - requireNonEmpty("addBoardSeat", "organizationName", input.organizationName); - const sourceUrl = requireSourceUrl("addBoardSeat", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO board_seats ( - id, entity_id, organization_entity_id, organization_name, role, - is_independent, committee, started_at, ended_at, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, organization_name, COALESCE(started_at,'')) DO UPDATE SET - organization_entity_id = COALESCE(excluded.organization_entity_id, board_seats.organization_entity_id), - role = COALESCE(excluded.role, board_seats.role), - is_independent = excluded.is_independent, - committee = COALESCE(excluded.committee, board_seats.committee), - ended_at = COALESCE(excluded.ended_at, board_seats.ended_at), - confidence = MAX(excluded.confidence, board_seats.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.role ?? null, input.isIndependent ? 1 : 0, input.committee ?? null, input.startedAt ?? null, input.endedAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.board_seat", - sourceUrl, - valueJson: { - organization_entity_id: input.organizationEntityId ?? null, - organization_name: input.organizationName, - role: input.role ?? null, - is_independent: input.isIndependent === true, - committee: input.committee ?? null, - started_at: input.startedAt ?? null, - ended_at: input.endedAt ?? null, - }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 4. addEducation — natural key (entity_id, institution, degree, ended_year). -// ========================================================================= -export async function addEducation(env, input) { - requireNonEmpty("addEducation", "entityId", input.entityId); - requireNonEmpty("addEducation", "institution", input.institution); - const sourceUrl = requireSourceUrl("addEducation", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO education_history ( - id, entity_id, institution, degree, field, started_year, ended_year, - honors, source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, institution, COALESCE(degree,''), COALESCE(ended_year,0)) DO UPDATE SET - field = COALESCE(excluded.field, education_history.field), - started_year = COALESCE(excluded.started_year, education_history.started_year), - honors = COALESCE(excluded.honors, education_history.honors), - confidence = MAX(excluded.confidence, education_history.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.institution, input.degree ?? null, input.field ?? null, input.startedYear ?? null, input.endedYear ?? null, input.honors ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.education", - sourceUrl, - valueJson: { - institution: input.institution, - degree: input.degree ?? null, - field: input.field ?? null, - started_year: input.startedYear ?? null, - ended_year: input.endedYear ?? null, - honors: input.honors ?? null, - }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 5. addFamilyTie — natural key (entity_id, relation_type, related_name). -// -// Privacy gate (task contract, "No PII leakage"): -// * `isPublic` is required — no defaulting (the helper throws if it's -// undefined / not a boolean), so a caller can never implicitly create -// a private row by omission. -// * `isPublic === false` additionally requires `isOperatorAsserted === -// true`. This is the explicit operator-intent marker the task calls -// out: only the human operator (via an authenticated route handler -// that sets this flag) can stash a private relationship. Background -// enrichment, agents, and scrapers will never have it set and will -// be rejected. -// * Private rows are NEVER mirrored into `facts` — `facts` is the -// public/agent retrieval surface, and routing PII through it would -// undermine row-level filtering downstream. The structured -// `family_ties` row is the only store for private ties, and the -// route layer is responsible for gating reads by operator identity. -// ========================================================================= -export async function addFamilyTie(env, input) { - requireNonEmpty("addFamilyTie", "entityId", input.entityId); - requireNonEmpty("addFamilyTie", "relationType", input.relationType); - requireNonEmpty("addFamilyTie", "relatedName", input.relatedName); - if (typeof input.isPublic !== "boolean") { - throw new Error("profile.addFamilyTie: isPublic is required and must be an explicit boolean"); - } - if (input.isPublic === false && input.isOperatorAsserted !== true) { - throw new Error("profile.addFamilyTie: private family ties (isPublic=false) require " + - "isOperatorAsserted=true; background enrichers and agents cannot store private relationships"); - } - const sourceUrl = requireSourceUrl("addFamilyTie", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO family_ties ( - id, entity_id, relation_type, related_name, related_entity_id, notes, - is_public, source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, relation_type, related_name) DO UPDATE SET - related_entity_id = COALESCE(excluded.related_entity_id, family_ties.related_entity_id), - notes = COALESCE(excluded.notes, family_ties.notes), - is_public = excluded.is_public, - confidence = MAX(excluded.confidence, family_ties.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.relationType, input.relatedName, input.relatedEntityId ?? null, input.notes ?? null, input.isPublic ? 1 : 0, sourceUrl, input.confidence ?? 1.0, now, now).run(); - // PII firewall: only public ties are projected to `facts`. The - // structured row above is still written for the operator-only UI. - if (input.isPublic === true) { - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.family_tie", - sourceUrl, - valueJson: { - relation_type: input.relationType, - related_name: input.relatedName, - related_entity_id: input.relatedEntityId ?? null, - notes: input.notes ?? null, - is_public: true, - }, - confidence: input.confidence, - observedAt: now, - }); - } - }); -} -// ========================================================================= -// 6. addPreference — upsert on (entity_id, preference_key). -// Mirrors to person.preference.{preferenceKey}; that dynamic predicate -// MUST exist in the registry (validated in mirrorFact + smoke test). -// ========================================================================= -export async function addPreference(env, input) { - requireNonEmpty("addPreference", "entityId", input.entityId); - requireNonEmpty("addPreference", "preferenceKey", input.preferenceKey); - const sourceUrl = requireSourceUrl("addPreference", input.sourceUrl); - const predicate = `person.preference.${input.preferenceKey}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO person_preferences ( - id, entity_id, preference_key, value_text, value_json, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, preference_key) DO UPDATE SET - value_text = COALESCE(excluded.value_text, person_preferences.value_text), - value_json = COALESCE(excluded.value_json, person_preferences.value_json), - source_url = excluded.source_url, - confidence = MAX(excluded.confidence, person_preferences.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.preferenceKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.valueText ?? null, - valueJson: input.valueJson ?? null, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 7. addInterest — natural key (entity_id, interest_category, interest_value). -// ========================================================================= -export async function addInterest(env, input) { - requireNonEmpty("addInterest", "entityId", input.entityId); - requireNonEmpty("addInterest", "interestCategory", input.interestCategory); - requireNonEmpty("addInterest", "interestValue", input.interestValue); - const sourceUrl = requireSourceUrl("addInterest", input.sourceUrl); - const predicate = `person.interest.${input.interestCategory}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO person_interests ( - id, entity_id, interest_category, interest_value, weight, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, interest_category, interest_value) DO UPDATE SET - weight = MAX(excluded.weight, person_interests.weight), - confidence = MAX(excluded.confidence, person_interests.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.interestCategory, input.interestValue, input.weight ?? 1.0, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.interestValue, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 8. addLifestyleSignal — natural key (entity_id, signal_key, observed_at). -// ========================================================================= -export async function addLifestyleSignal(env, input) { - requireNonEmpty("addLifestyleSignal", "entityId", input.entityId); - requireNonEmpty("addLifestyleSignal", "signalKey", input.signalKey); - const sourceUrl = requireSourceUrl("addLifestyleSignal", input.sourceUrl); - const predicate = `person.lifestyle.${input.signalKey}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO lifestyle_signals ( - id, entity_id, signal_key, value_text, value_json, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, signal_key) DO UPDATE SET - value_text = COALESCE(excluded.value_text, lifestyle_signals.value_text), - value_json = COALESCE(excluded.value_json, lifestyle_signals.value_json), - source_url = excluded.source_url, - confidence = MAX(excluded.confidence, lifestyle_signals.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.valueText ?? null, - valueJson: input.valueJson ?? null, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 9. addTravelPattern — natural key (entity_id, pattern_kind, place, starts_at). -// ========================================================================= -export async function addTravelPattern(env, input) { - requireNonEmpty("addTravelPattern", "entityId", input.entityId); - requireNonEmpty("addTravelPattern", "patternKind", input.patternKind); - requireNonEmpty("addTravelPattern", "place", input.place); - const sourceUrl = requireSourceUrl("addTravelPattern", input.sourceUrl); - const predicate = `person.travel.${input.patternKind}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO travel_patterns ( - id, entity_id, pattern_kind, place, country_iso2, starts_at, ends_at, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, pattern_kind, place, COALESCE(starts_at,'')) DO UPDATE SET - country_iso2 = COALESCE(excluded.country_iso2, travel_patterns.country_iso2), - ends_at = COALESCE(excluded.ends_at, travel_patterns.ends_at), - confidence = MAX(excluded.confidence, travel_patterns.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.patternKind, input.place, input.countryIso2 ?? null, input.startsAt ?? null, input.endsAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.place, - valueJson: { - country_iso2: input.countryIso2 ?? null, - starts_at: input.startsAt ?? null, - ends_at: input.endsAt ?? null, - }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 10. addConferenceAttendance — UNIQUE(entity_id, conference_name, year). -// ========================================================================= -export async function addConferenceAttendance(env, input) { - requireNonEmpty("addConferenceAttendance", "entityId", input.entityId); - requireNonEmpty("addConferenceAttendance", "conferenceName", input.conferenceName); - if (!Number.isInteger(input.year) || input.year < 1900 || input.year > 2100) { - throw new Error("profile.addConferenceAttendance: year must be a 4-digit integer"); - } - const sourceUrl = requireSourceUrl("addConferenceAttendance", input.sourceUrl); - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO conference_attendance ( - id, entity_id, conference_name, year, role, session_topic, city, country_iso2, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, conference_name, year) DO UPDATE SET - role = COALESCE(excluded.role, conference_attendance.role), - session_topic = COALESCE(excluded.session_topic, conference_attendance.session_topic), - city = COALESCE(excluded.city, conference_attendance.city), - country_iso2 = COALESCE(excluded.country_iso2, conference_attendance.country_iso2), - confidence = MAX(excluded.confidence, conference_attendance.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.conferenceName, input.year, input.role ?? null, input.sessionTopic ?? null, input.city ?? null, input.countryIso2 ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate: "person.conference", - sourceUrl, - valueJson: { - conference_name: input.conferenceName, - year: input.year, - role: input.role ?? null, - session_topic: input.sessionTopic ?? null, - city: input.city ?? null, - country_iso2: input.countryIso2 ?? null, - }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 11. addGoal — natural key (entity_id, goal_kind, goal_text). -// ========================================================================= -export async function addGoal(env, input) { - requireNonEmpty("addGoal", "entityId", input.entityId); - requireNonEmpty("addGoal", "goalKind", input.goalKind); - requireNonEmpty("addGoal", "goalText", input.goalText); - const sourceUrl = requireSourceUrl("addGoal", input.sourceUrl); - const predicate = `person.goal.${input.goalKind}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO person_goals ( - id, entity_id, goal_kind, goal_text, target_date, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, goal_kind, goal_text) DO UPDATE SET - target_date = COALESCE(excluded.target_date, person_goals.target_date), - confidence = MAX(excluded.confidence, person_goals.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.goalKind, input.goalText, input.targetDate ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.goalText, - valueJson: { target_date: input.targetDate ?? null }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 12. addConversationHook — natural key (entity_id, hook_kind, hook_text). -// ========================================================================= -export async function addConversationHook(env, input) { - requireNonEmpty("addConversationHook", "entityId", input.entityId); - requireNonEmpty("addConversationHook", "hookKind", input.hookKind); - requireNonEmpty("addConversationHook", "hookText", input.hookText); - const sourceUrl = requireSourceUrl("addConversationHook", input.sourceUrl); - const predicate = `person.hook.${input.hookKind}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO conversation_hooks ( - id, entity_id, hook_kind, hook_text, related_entity_id, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, hook_kind, hook_text) DO UPDATE SET - related_entity_id = COALESCE(excluded.related_entity_id, conversation_hooks.related_entity_id), - confidence = MAX(excluded.confidence, conversation_hooks.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.hookKind, input.hookText, input.relatedEntityId ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.hookText, - valueJson: { related_entity_id: input.relatedEntityId ?? null }, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// ========================================================================= -// 13. addAppreciationSignal — natural key (entity_id, signal_kind, signal_text). -// ========================================================================= -export async function addAppreciationSignal(env, input) { - requireNonEmpty("addAppreciationSignal", "entityId", input.entityId); - requireNonEmpty("addAppreciationSignal", "signalKind", input.signalKind); - requireNonEmpty("addAppreciationSignal", "signalText", input.signalText); - const sourceUrl = requireSourceUrl("addAppreciationSignal", input.sourceUrl); - const predicate = `person.appreciation.${input.signalKind}`; - const now = input.observedAt ?? nowIso(); - await withProfileLock(env, input.entityId, async () => { - await env.DB.prepare(`INSERT INTO appreciation_signals ( - id, entity_id, signal_kind, signal_text, - source_url, confidence, observed_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id, signal_kind, signal_text) DO UPDATE SET - confidence = MAX(excluded.confidence, appreciation_signals.confidence), - observed_at = excluded.observed_at, - updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKind, input.signalText, sourceUrl, input.confidence ?? 1.0, now, now).run(); - await mirrorFact(env, { - entityId: input.entityId, - predicate, - sourceUrl, - valueText: input.signalText, - confidence: input.confidence, - observedAt: now, - }); - }); -} -// EntityService facade — single import surface for callers. -export const EntityService = { - setPersonIdentity, - addCareerEntry, - addBoardSeat, - addEducation, - addFamilyTie, - addPreference, - addInterest, - addLifestyleSignal, - addTravelPattern, - addConferenceAttendance, - addGoal, - addConversationHook, - addAppreciationSignal, -}; -export { EMITTED_PREDICATES }; diff --git a/apps/worker/test-dist-q/entities/query.js b/apps/worker/test-dist-q/entities/query.js deleted file mode 100644 index 79f93c4f..00000000 --- a/apps/worker/test-dist-q/entities/query.js +++ /dev/null @@ -1,154 +0,0 @@ -// High-level read helpers. `loadEntity` returns the canonical envelope -// every consumer can rely on; `searchEntities` is a thin filter DSL that -// joins entity_summary + entity_tags for sub-50ms list responses. -import { getEffectiveFacts, loadCurrentOverrides } from "./facts"; -export async function loadEntity(env, id, opts) { - const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first(); - if (!ent) - return null; - // Task #3 (Editable Profiles): use the SAME shared resolver as the - // summary rebuilder so the two read sites cannot drift. The resolver - // returns one EffectiveFact list with canonical rows + dethroned - // attempts marked overridden_attempt=true; we split that into - // facts[] (canonical) and attempts[] (diff strip). - const [roles, effective, channels, tags, summary, overridesMap] = await Promise.all([ - env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(id).all(), - getEffectiveFacts(env, id, { includeNonCurrent: !!opts?.includeNonCurrent, limit: 500 }), - env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified, is_dnc FROM channels WHERE entity_id = ?`).bind(id).all(), - env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(id).all(), - env.DB.prepare(`SELECT * FROM entity_summary WHERE entity_id = ?`).bind(id).first(), - loadCurrentOverrides(env, id), - ]); - const factRows = []; - const attempts = []; - for (const e of effective) { - const row = { - id: e.id, predicate: e.predicate, - value_text: e.value_text, value_number: e.value_number, - value_json: e.value_json, value_entity_id: e.value_entity_id, - source: e.source, source_kind: e.source_kind, - confidence: e.confidence, verified_score: e.verified_score, - observed_at: e.observed_at, is_current: e.is_current, - superseded_by_override: e.superseded_by_override, - }; - if (e.overridden_attempt) - attempts.push(row); - else - factRows.push(row); - } - const overrideArr = Array.from(overridesMap.values()).map((ov) => ({ - id: ov.id, - predicate: ov.predicate, - value_text: ov.value_text, - value_number: ov.value_numeric, - value_json: ov.value_json ? (() => { try { - return JSON.parse(ov.value_json); - } - catch { - return ov.value_json; - } })() : null, - overridden_at: ov.overridden_at, - })); - return { - id, kind: ent.kind, entity: ent, - roles: (roles.results ?? []), - facts: factRows, - attempts, - channels: (channels.results ?? []), - tags: (tags.results ?? []), - summary: summary ?? null, - overrides: overrideArr, - }; -} -export async function searchEntities(env, f) { - // IMPORTANT: bind order must match SQL placeholder order. Since the - // tag JOINs appear *before* the WHERE clause in the final SQL, their - // binds must be pushed first. We build two separate bind arrays and - // concatenate them in SQL order at the end. - const joinBinds = []; - const whereBinds = []; - const where = ["s.status = 'active'"]; - if (f.kind) { - where.push("s.kind = ?"); - whereBinds.push(f.kind); - } - if (f.role) { - where.push("s.primary_role = ?"); - whereBinds.push(f.role); - } - if (f.country_iso2) { - where.push("s.country_iso2 = ?"); - whereBinds.push(f.country_iso2.toUpperCase()); - } - if (typeof f.check_min_usd === "number") { - where.push("s.check_size_max_usd >= ?"); - whereBinds.push(f.check_min_usd); - } - if (typeof f.check_max_usd === "number") { - where.push("s.check_size_min_usd <= ?"); - whereBinds.push(f.check_max_usd); - } - if (f.has_unicorn) { - where.push("s.unicorn_count > 0"); - } - if (typeof f.min_fit === "number") { - where.push("s.fit_max_score >= ?"); - whereBinds.push(f.min_fit); - } - if (typeof f.min_intent === "number") { - where.push("s.intent_score >= ?"); - whereBinds.push(f.min_intent); - } - if (f.q) { - where.push("(lower(s.display_name) LIKE ? OR lower(s.primary_domain) LIKE ? OR lower(s.primary_email) LIKE ?)"); - const q = `%${f.q.toLowerCase()}%`; - whereBinds.push(q, q, q); - } - // Tag filters require a JOIN per taxonomy so we can intersect. - // NOTE: `has_role` is *not* a tag — roles live in entity_roles - // (addRole writes there, not entity_tags). The previous JOIN on - // entity_tags taxonomy='role' returned empty results for every - // wrapper helper (listFirms, listInvestors, listCompanies, - // listAccounts, listBuyers, listFounders). We now JOIN entity_roles - // directly for role membership. - const joins = []; - let tagJoinIdx = 0; - for (const [tax, slug] of [["sector", f.sector], ["stage", f.stage], ["geo", f.geo]]) { - if (!slug) - continue; - const alias = `t${++tagJoinIdx}`; - joins.push(`JOIN entity_tags ${alias} ON ${alias}.entity_id = s.entity_id AND ${alias}.taxonomy = ? AND ${alias}.slug = ?`); - joinBinds.push(tax, slug); - } - if (f.has_role) { - joins.push(`JOIN entity_roles er ON er.entity_id = s.entity_id AND er.role = ?`); - joinBinds.push(f.has_role); - } - const sortCol = (() => { - switch (f.sort) { - case "intent": return "s.intent_score DESC"; - case "quality": return "s.quality_score DESC"; - case "updated": return "s.rebuilt_at DESC"; - default: return "s.fit_max_score DESC"; - } - })(); - const limit = Math.min(Math.max(1, f.limit ?? 50), 200); - const offset = Math.max(0, f.offset ?? 0); - const sql = `SELECT s.* FROM entity_summary s ${joins.join(" ")} - WHERE ${where.join(" AND ")} - ORDER BY ${sortCol}, s.entity_id ASC - LIMIT ? OFFSET ?`; - const r = await env.DB.prepare(sql).bind(...joinBinds, ...whereBinds, limit + 1, offset).all(); - const rows = r.results ?? []; - const hasMore = rows.length > limit; - return { - items: (hasMore ? rows.slice(0, limit) : rows), - next_offset: hasMore ? offset + limit : null, - }; -} -export const listFirms = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "firm" }); -export const listInvestors = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "investor" }); -export const listCompanies = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "company" }); -export const listAccounts = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: "account" }); -export const listBuyers = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "buyer" }); -export const listFounders = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "founder" }); diff --git a/apps/worker/test-dist-q/entities/roles.js b/apps/worker/test-dist-q/entities/roles.js deleted file mode 100644 index 2b16577f..00000000 --- a/apps/worker/test-dist-q/entities/roles.js +++ /dev/null @@ -1,120 +0,0 @@ -import { isGarbage, logDataQuality, classifyPersonName } from "./garbage"; -export async function createEntity(env, init) { - // Task #9: pre-insert garbage guard. The pure heuristic detector - // (no AI call here — keep createEntity synchronous and cheap on the - // hot write path) rejects HTML page titles / nav strings / UI - // labels that the crawler may have mistaken for entity names. - // Returns null + audit row instead of throwing so callers can - // skip without crashing the broader import. The AI second opinion - // runs only in the cron sweep, NOT inline on every write. - const verdict = isGarbage({ - kind: init.kind, - display_name: init.display_name ?? null, - primary_url: init.primary_url ?? null, - primary_domain: init.primary_domain ?? null, - primary_email_key: init.primary_email_key ?? null, - primary_linkedin_key: init.primary_linkedin_key ?? null, - }); - if (verdict.is_garbage) { - console.log("garbage.pre_insert_rejected", JSON.stringify({ - kind: init.kind, display_name: init.display_name, reasons: verdict.reasons, - })); - // Log to data_quality_log with a synthetic entity_id so operators - // can audit rejected writes too. We use a `rejected:` prefix to - // distinguish from soft-deleted entities (which carry real ids). - void logDataQuality(env, "rejected:" + (init.display_name ?? "").slice(0, 100), "pre_insert_rejected", verdict.reasons, "pre_insert_guard", null).catch(() => undefined); - return null; - } - // Task #6: reclassify-on-write. A `person` whose display name is clearly - // an organization ("Intel Capital", "Mendoza Ventures") is written as an - // `org` so it never lands in the People list in the first place. A strong - // personal identifier (personal LinkedIn /in/ or email) contradicting the - // org-suffix name suppresses the flip — never mislabel a likely real - // person. Junk names were already rejected by the isGarbage guard above. - let effectiveKind = init.kind; - let reclassifiedOrgRole = null; - if (init.kind === "person") { - const cls = classifyPersonName(init.display_name ?? null); - if (cls.verdict === "organization" && cls.orgRole) { - const personalLinkedin = !!init.primary_linkedin_key && /(^|\/)in\//i.test(init.primary_linkedin_key); - const hasEmail = !!init.primary_email_key; - if (!personalLinkedin && !hasEmail) { - effectiveKind = "org"; - reclassifiedOrgRole = cls.orgRole; - } - } - } - const id = crypto.randomUUID(); - const now = new Date().toISOString(); - await env.DB.prepare(`INSERT INTO u_entities ( - id, kind, display_name, primary_url, primary_domain, - primary_email_key, primary_linkedin_key, primary_twitter_handle, primary_github_handle, - status, created_at, updated_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'active', ?, ?)`).bind(id, effectiveKind, init.display_name ?? null, init.primary_url ?? null, init.primary_domain ?? null, init.primary_email_key ?? null, init.primary_linkedin_key ?? null, init.primary_twitter_handle ?? null, init.primary_github_handle ?? null, now, now).run(); - await env.DB.prepare(`INSERT INTO entity_history (id, entity_id, action, source, changed_at) VALUES (?, ?, 'create', 'system', ?)`).bind(crypto.randomUUID(), id, now).run(); - // Task #8: enqueue persona ↔ entity matching for newly-created - // person entities so a freshly-created founder/operator appears in - // matching personas' candidate lists within minutes — even before - // any career/title facts are written. KV-debounced inside trigger. - if (effectiveKind === "person") { - try { - const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); - void triggerEntityMatchRefresh(env, id).catch(() => undefined); - } - catch { /* best-effort */ } - } - // Task #6: stamp the inferred org role + an audit row when a person was - // reclassified to an org on write, so the operator console can trace it. - if (reclassifiedOrgRole) { - await addRole(env, id, reclassifiedOrgRole, { is_primary: true, source: "garbage_reclassify_on_write" }); - void logDataQuality(env, id, "reclassified", [`org_role:${reclassifiedOrgRole}`, "reclassified_on_write"], "pre_insert_guard", null).catch(() => undefined); - } - // Task #3 (AI Profile Filler): auto-trigger a profile fill for newly - // created org entities that have a website but no facts yet (the - // signal-poor "low confidence" case the spec calls out). Dispatched - // via WF binding when available so the cost is async and respects - // the daily neuron cap. No-op when the binding isn't configured. - if (effectiveKind === "org" && (init.primary_url || init.primary_domain) && !init.suppressAutoProfileFill) { - const wf = env.WF_PROFILE_FILLER; - if (wf) { - try { - void wf.create({ params: { entityId: id, force: false, triggeredBy: "auto:entity_created" } }).catch(() => undefined); - } - catch { /* best-effort */ } - } - } - // Task #4 (Relationship Inference Worker): debounced enqueue into - // relationship_infer_queue (migration 377). The consolidated nightly - // slot drains the queue with the per-entity orchestrator pass. - try { - const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); - void enqueueRelInfer(env, id, `created:${effectiveKind}`).catch(() => undefined); - } - catch { /* best-effort */ } - return (await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first()); -} -export async function addRole(env, entityId, role, opts) { - try { - await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) - VALUES (?, ?, ?, ?, ?) - ON CONFLICT(entity_id, role) DO UPDATE SET - is_primary = MAX(is_primary, excluded.is_primary), - confidence = MAX(confidence, excluded.confidence)`).bind(entityId, role, opts?.is_primary ? 1 : 0, opts?.source ?? null, opts?.confidence ?? 1).run(); - } - catch (e) { - console.warn("addRole failed", role, e.message); - } -} -export async function getLegacyEntityId(env, table, legacyId) { - const r = await env.DB.prepare(`SELECT entity_id FROM entity_legacy_map WHERE legacy_table = ? AND legacy_id = ?`).bind(table, String(legacyId)).first(); - return r?.entity_id ?? null; -} -export async function setLegacyEntityId(env, table, legacyId, entityId) { - try { - await env.DB.prepare(`INSERT INTO entity_legacy_map (legacy_table, legacy_id, entity_id) - VALUES (?, ?, ?) ON CONFLICT DO NOTHING`).bind(table, String(legacyId), entityId).run(); - } - catch (e) { - console.warn("setLegacyEntityId failed", table, legacyId, e.message); - } -} diff --git a/apps/worker/test-dist-q/entities/summary.js b/apps/worker/test-dist-q/entities/summary.js deleted file mode 100644 index fc6eef0b..00000000 --- a/apps/worker/test-dist-q/entities/summary.js +++ /dev/null @@ -1,121 +0,0 @@ -// Rebuild `entity_summary` for one entity from its current facts + -// channels + tags + roles. Runs inside the queue consumer. -import { getEffectiveFacts } from "./facts"; -const SOURCE_PRIORITY = { - manual: 5, enrichment: 4, import: 3, scrape: 2, ai: 1, inferred: 0, -}; -function pickBestFact(rows) { - if (!rows.length) - return null; - return rows.slice().sort((a, b) => { - const sa = (a.confidence ?? 0) * 100 + (SOURCE_PRIORITY[a.source_kind] ?? 0) * 10 + Date.parse(a.observed_at) / 1e12; - const sb = (b.confidence ?? 0) * 100 + (SOURCE_PRIORITY[b.source_kind] ?? 0) * 10 + Date.parse(b.observed_at) / 1e12; - return sb - sa; - })[0]; -} -function txt(rows, predicate) { - const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); - return f?.value_text ?? null; -} -function num(rows, predicate) { - const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); - return f?.value_number ?? null; -} -export async function rebuildSummary(env, entityId) { - const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(entityId).first(); - if (!ent) - return false; - if (ent.status === "merged" || ent.status === "soft_deleted") { - await env.DB.prepare(`DELETE FROM entity_summary WHERE entity_id = ?`).bind(entityId).run(); - return true; - } - // Task #3 (Editable Profiles): the summary input is the EFFECTIVE - // facts view — overrides win, overridden_attempt rows are filtered. - // Same resolver as the per-entity read path in query.ts, so the two - // call sites cannot drift. - const [effective, tagsRes, rolesRes, channelsRes] = await Promise.all([ - getEffectiveFacts(env, entityId), - env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(entityId).all(), - env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(), - env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified FROM channels WHERE entity_id = ?`).bind(entityId).all(), - ]); - const facts = effective - .filter((e) => !e.overridden_attempt) - .map((e) => ({ - predicate: e.predicate, - value_text: e.value_text, - value_number: e.value_number, - value_json: e.value_json != null ? (typeof e.value_json === "string" ? e.value_json : JSON.stringify(e.value_json)) : null, - value_entity_id: e.value_entity_id, - confidence: e.confidence, - observed_at: e.observed_at, - source_kind: e.source_kind, - })); - const tags = tagsRes.results ?? []; - const roles = rolesRes.results ?? []; - const channels = channelsRes.results ?? []; - const primaryRole = (roles.find((r) => r.is_primary === 1) ?? roles[0])?.role ?? null; - const display = ent.display_name ?? txt(facts, "name") ?? txt(facts, "display_name"); - const country = txt(facts, "country_iso2"); - const region = txt(facts, "region"); - const city = txt(facts, "city"); - const sectors = tags.filter((t) => t.taxonomy === "sector").map((t) => t.slug); - const stages = tags.filter((t) => t.taxonomy === "stage").map((t) => t.slug); - const geos = tags.filter((t) => t.taxonomy === "geo").map((t) => t.slug); - const checkMin = num(facts, "check_size_min_usd"); - const checkMax = num(facts, "check_size_max_usd"); - const fitMax = num(facts, "fit_max_score") ?? 0; - const intent = num(facts, "intent_score") ?? 0; - const unicornCount = Math.round(num(facts, "unicorn_count") ?? 0); - const primaryEmailChan = channels.filter((c) => c.kind === "email").sort((a, b) => (b.is_primary - a.is_primary) || (b.is_verified - a.is_verified))[0]; - const primaryLinkedinChan = channels.filter((c) => c.kind === "linkedin")[0]; - const employerEntityId = (facts.find((f) => f.predicate === "employer" && f.value_entity_id) ?? null)?.value_entity_id ?? null; - let employerDisplay = null; - if (employerEntityId) { - const r = await env.DB.prepare(`SELECT display_name FROM u_entities WHERE id = ?`).bind(employerEntityId).first(); - employerDisplay = r?.display_name ?? null; - } - else { - employerDisplay = txt(facts, "primary_employer") ?? txt(facts, "org"); - } - // Quality: coverage × confidence average × source diversity, scaled 0..100. - const coverage = Math.min(1, facts.length / 12); - const confAvg = facts.length ? facts.reduce((s, f) => s + (f.confidence ?? 0), 0) / facts.length : 0; - const diversity = Math.min(1, new Set(facts.map((f) => f.source_kind)).size / 3); - const quality = Math.round((coverage * 0.5 + confAvg * 0.3 + diversity * 0.2) * 100); - const now = new Date().toISOString(); - await env.DB.prepare(`INSERT INTO entity_summary ( - entity_id, kind, display_name, primary_role, primary_employer, primary_employer_entity_id, - country_iso2, region, city, sectors_csv, stages_csv, geos_csv, - check_size_min_usd, check_size_max_usd, primary_email, primary_linkedin, - primary_domain, quality_score, fit_max_score, intent_score, unicorn_count, status, rebuilt_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(entity_id) DO UPDATE SET - kind = excluded.kind, - display_name = excluded.display_name, - primary_role = excluded.primary_role, - primary_employer = excluded.primary_employer, - primary_employer_entity_id = excluded.primary_employer_entity_id, - country_iso2 = excluded.country_iso2, - region = excluded.region, - city = excluded.city, - sectors_csv = excluded.sectors_csv, - stages_csv = excluded.stages_csv, - geos_csv = excluded.geos_csv, - check_size_min_usd = excluded.check_size_min_usd, - check_size_max_usd = excluded.check_size_max_usd, - primary_email = excluded.primary_email, - primary_linkedin = excluded.primary_linkedin, - primary_domain = excluded.primary_domain, - quality_score = excluded.quality_score, - fit_max_score = excluded.fit_max_score, - intent_score = excluded.intent_score, - unicorn_count = excluded.unicorn_count, - status = excluded.status, - rebuilt_at = excluded.rebuilt_at`).bind(entityId, ent.kind, display, primaryRole, employerDisplay, employerEntityId, country, region, city, sectors.join(","), stages.join(","), geos.join(","), checkMin != null ? Math.round(checkMin) : null, checkMax != null ? Math.round(checkMax) : null, primaryEmailChan?.canonical ?? null, primaryLinkedinChan?.canonical ?? null, ent.primary_domain, quality, fitMax, intent, unicornCount, ent.status, now).run(); - // Update u_entities.quality_score + last_summary_at so list paths that - // still read from `u_entities` see the latest score. - await env.DB.prepare(`UPDATE u_entities SET quality_score = ?, last_summary_at = ?, updated_at = ? WHERE id = ?`) - .bind(quality, now, now, entityId).run(); - return true; -} diff --git a/apps/worker/test-dist-q/entities/summaryQueue.js b/apps/worker/test-dist-q/entities/summaryQueue.js deleted file mode 100644 index ccbf3c4b..00000000 --- a/apps/worker/test-dist-q/entities/summaryQueue.js +++ /dev/null @@ -1,26 +0,0 @@ -// Enqueue + consume `rebuild_summary` work. Uses the existing LEAD_QUEUE -// to avoid provisioning a second queue; the consumer in index.ts -// dispatches by message shape. -import { rebuildSummary } from "./summary"; -export function isRebuildSummaryMessage(m) { - return !!m && typeof m === "object" && m.type === "rebuild_summary" - && typeof m.entityId === "string"; -} -// We intentionally do *not* debounce via KV: the previous design dropped -// later writes when a debounce key was already set, which left -// entity_summary stale until the next mutation. The queue handler can -// coalesce duplicates at consume time if needed; in the meantime, the -// extra work is one upsert into a tiny rollup table. -export async function enqueueSummaryRebuild(env, entityId) { - if (!entityId) - return; - try { - await env.LEAD_QUEUE.send({ type: "rebuild_summary", entityId }); - } - catch (e) { - console.warn("enqueueSummaryRebuild failed", entityId, e.message); - } -} -export async function handleSummaryMessage(env, m) { - await rebuildSummary(env, m.entityId); -} diff --git a/apps/worker/test-dist-q/entities/tags.js b/apps/worker/test-dist-q/entities/tags.js deleted file mode 100644 index a28e6fe3..00000000 --- a/apps/worker/test-dist-q/entities/tags.js +++ /dev/null @@ -1,37 +0,0 @@ -export async function addTag(env, t) { - if (!t.entity_id || !t.slug) - return; - const slug = String(t.slug).trim().toLowerCase(); - if (!slug) - return; - try { - await env.DB.prepare(`INSERT INTO entity_tags (entity_id, taxonomy, slug, weight, source) - VALUES (?, ?, ?, ?, ?) - ON CONFLICT(entity_id, taxonomy, slug) - DO UPDATE SET weight = MAX(weight, excluded.weight)`).bind(t.entity_id, t.taxonomy, slug, t.weight ?? 1, t.source ?? null).run(); - } - catch (e) { - console.warn("addTag failed", t.taxonomy, slug, e.message); - } -} -export async function addTagsFromJsonArray(env, entityId, taxonomy, rawJsonArray, source) { - if (!rawJsonArray) - return 0; - let arr; - try { - arr = JSON.parse(rawJsonArray); - } - catch { - return 0; - } - if (!Array.isArray(arr)) - return 0; - let n = 0; - for (const v of arr) { - if (typeof v !== "string") - continue; - await addTag(env, { entity_id: entityId, taxonomy, slug: v, source }); - n += 1; - } - return n; -} diff --git a/apps/worker/test-dist-q/errors.js b/apps/worker/test-dist-q/errors.js deleted file mode 100644 index 653c1b1a..00000000 --- a/apps/worker/test-dist-q/errors.js +++ /dev/null @@ -1,328 +0,0 @@ -// Centralized error taxonomy for the worker (Task #27). -// -// Every operational failure should be either: -// 1. an `AppError` subclass thrown explicitly, OR -// 2. caught and re-thrown via `wrapUnknown(e, code, ctx)` so the global -// onError handler in `index.ts` can serialize it to JSON, log it to -// `error_log`, mirror it to Analytics Engine, and attach a request_id. -// -// All AppErrors carry: -// - code: stable machine string from the ErrCode union below. -// - status: HTTP status to return when surfaced via API. -// - kind: high-level category for UI grouping. -// - retryable: hint to the queue/retry layer. -// - context: free-form structured context (job_id, url, host, provider…). -export class AppError extends Error { - code; - kind; - status; - retryable; - context; - cause; - constructor(opts) { - super(opts.message ?? opts.code); - this.name = "AppError"; - this.code = opts.code; - this.kind = opts.kind; - this.status = opts.status ?? defaultStatusForKind(opts.kind); - this.retryable = opts.retryable ?? defaultRetryableForKind(opts.kind); - this.context = opts.context ?? {}; - if (opts.cause instanceof Error) - this.cause = opts.cause; - } - toJSON(requestId) { - const out = { - error: this.code, - code: this.code, - kind: this.kind, - status: this.status, - message: this.message, - retryable: this.retryable, - }; - if (Object.keys(this.context).length) - out.context = this.context; - if (requestId) - out.request_id = requestId; - if (this.cause) { - const c = { - name: this.cause.name, - message: this.cause.message, - }; - if (this.cause.stack) - c.stack = this.cause.stack; - out.cause = c; - } - return out; - } -} -function defaultStatusForKind(kind) { - switch (kind) { - case "validation": return 400; - case "auth": return 401; - case "permanent": return 422; - case "config": return 500; - case "upstream": return 502; - case "transient": return 503; - // Task #72: a benign skip is not an error surface — neutral 200. - case "skip": return 200; - case "internal": - default: return 500; - } -} -function defaultRetryableForKind(kind) { - return kind === "transient" || kind === "upstream"; -} -// ---- Common subclasses (sugar; AppError directly is also fine) ---------- -export class ValidationError extends AppError { - constructor(code, message, context) { - super({ code, kind: "validation", status: 400, message, retryable: false, ...(context ? { context } : {}) }); - this.name = "ValidationError"; - } -} -export class NotFoundError extends AppError { - constructor(resource, id) { - super({ - code: "not_found", - kind: "permanent", - status: 404, - message: `${resource}${id ? ` ${id}` : ""} not found`, - retryable: false, - context: id ? { resource, id } : { resource }, - }); - this.name = "NotFoundError"; - } -} -export class AuthError extends AppError { - constructor(code, message, context) { - super({ - code, - kind: "auth", - status: code === "forbidden" ? 403 : 401, - message: message ?? code, - retryable: false, - ...(context ? { context } : {}), - }); - this.name = "AuthError"; - } -} -export class UpstreamError extends AppError { - constructor(provider, message, context) { - // Constructed at runtime; the closed ErrCode enum already enumerates - // every known provider so this assertion is the only escape. - const code = `upstream_${provider}`; - super({ - code, - kind: "upstream", - status: 502, - message, - retryable: true, - context: { provider, ...(context ?? {}) }, - }); - this.name = "UpstreamError"; - } -} -export class ScrapeBlockedError extends AppError { - constructor(host, reason, context) { - super({ - code: "scrape_blocked", - kind: "permanent", - status: 403, - message: `${host}: ${reason}`, - retryable: false, - context: { host, reason, ...(context ?? {}) }, - }); - this.name = "ScrapeBlockedError"; - } -} -export class BudgetExhaustedError extends AppError { - constructor(scope, context) { - super({ - code: "budget_exhausted", - kind: "permanent", - status: 429, - message: `Budget exhausted for ${scope}`, - retryable: false, - context: { scope, ...(context ?? {}) }, - }); - this.name = "BudgetExhaustedError"; - } -} -export class TransientError extends AppError { - constructor(code, message, context) { - super({ code, kind: "transient", status: 503, message, retryable: true, ...(context ? { context } : {}) }); - this.name = "TransientError"; - } -} -// ---- Wrappers -------------------------------------------------------------- -/** Convert any unknown thrown value into an AppError (idempotent). */ -export function wrapUnknown(e, code, context) { - if (e instanceof AppError) { - if (context) - Object.assign(e.context, context); - return e; - } - const err = e instanceof Error ? e : new Error(typeof e === "string" ? e : safeJson(e)); - // Heuristic upgrade: detect common transient patterns from the cause. - const guessed = classify(err); - const opts = { - code: guessed?.code ?? code, - kind: guessed?.kind ?? "internal", - message: err.message || code, - cause: err, - }; - if (guessed) - opts.retryable = guessed.retryable; - if (context) - opts.context = context; - return new AppError(opts); -} -/** Type guard. */ -export function isAppError(e) { - return e instanceof AppError; -} -/** - * Heuristic classifier for stringly-typed errors thrown by the existing - * codebase or by the platform (D1, fetch, Workers AI). Returns null if no - * pattern matches; callers should then use the supplied default code/kind. - * - * Required by Task #27 acceptance: every logged failure has a well-typed - * code, even when thrown deep in legacy code that hasn't migrated to - * AppError yet. - */ -export function classify(err) { - if (err instanceof AppError) - return { code: err.code, kind: err.kind, retryable: err.retryable }; - const msg = (err instanceof Error ? err.message : String(err ?? "")).toLowerCase(); - if (!msg) - return null; - // Task #70: Cloudflare's per-invocation subrequest cap surfaces as - // "Too many subrequests by single Worker invocation", which bubbles up - // wrapped in a `fetch_failed:proxy_error:...` (or `fetch_error:...`) - // string. This is NOT a permanent scrape block — the page is fine, the - // invocation just ran out of budget — so it must be classified - // transient/retryable AHEAD of the generic `fetch_failed:` permanent - // rule below. Matched specifically (not all proxy_errors) so genuine - // upstream proxy failures still dead-letter as before. - // - // `subrequest_budget_exhausted` is our OWN pre-emptive refusal (the - // crawl-path budget stopped a fetch before it could trip the platform - // cap); it is the same condition and must retry identically. - if (msg.includes("too many subrequests") || msg.includes("subrequest_budget")) { - return { code: "subrequest_limit", kind: "transient", retryable: true }; - } - // Pipeline-level fetch/scrape sentinels. These reasons are bubbled up - // from the scraper as plain `Error("fetch_failed::status=")` - // (see scraper/pipeline.ts). They are expected operational outcomes — - // not real internal errors — so we map them to typed codes the queue - // can dead-letter without paging. - if (msg.includes("scraping_api_not_configured") || - msg.includes("proxy_not_configured") || - msg.includes("browser_binding_unavailable") || - msg.includes("puppeteer_module_missing")) { - return { code: "config_missing", kind: "config", retryable: false }; - } - // Task #72: robots.txt / ToS blocks are EXPECTED, benign policy outcomes, - // not internal errors — honoring a host's robots.txt is correct behavior. - // Classify them as the `skip` kind so the queue routes them to the `skipped` - // terminal status (no error_log row, never retried) instead of failing / - // dead-lettering them as scrape errors. NB: the scraper emits the token - // `robots_disallow` (see scraper/robots.ts); the old code only matched the - // `robots_disallowed` spelling and silently fell through to the generic - // fetch_failed → permanent rule, so the block surfaced as a red 422. - if (msg.includes("robots_disallow") || msg.includes("tos_blocked")) { - const code = msg.includes("tos_blocked") ? "tos_blocked" : "robots_disallowed"; - return { code, kind: "skip", retryable: false }; - } - // Gated sources still need an operator manual-paste; that's a permanent - // scrape block (the queue preflight already skips it earlier — this is the - // fetcher backstop, kept permanent so its behavior is unchanged). - if (msg.includes("gated_source_use_manual_paste")) { - return { code: "scrape_blocked", kind: "permanent", retryable: false }; - } - if (msg.includes("no_table_found")) { - return { code: "parse_error", kind: "validation", retryable: false }; - } - if (msg.startsWith("fetch_failed:") || msg.includes(":fetch_failed:")) { - // Task #71: a `fetch_failed:` message can carry an embedded upstream HTTP - // status (e.g. "fetch_failed:status_429:status=429"). A 429 rate-limit and - // any 5xx are TRANSIENT — the page is fine, the upstream is just briefly - // refusing — so they must retry with backoff, not get dropped as a - // permanent scrape block on attempt 1. Parse the embedded status BEFORE the - // generic permanent fallback. A genuine 4xx (403/404/...) or a - // fetch_failed with no recoverable status still resolves to permanent - // scrape_blocked (prior behavior preserved). The dedicated scrape sentinels - // (robots/tos/gated/config) matched above stay permanent regardless. - const embedded = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); - if (embedded) { - const s = Number(embedded[1]); - if (s === 429) - return { code: "rate_limited", kind: "transient", retryable: true }; - if (s >= 500) - return { code: "fetch.http_5xx", kind: "transient", retryable: true }; - } - // Generic fetch_failed without a recoverable status → upstream/permanent. - return { code: "scrape_blocked", kind: "permanent", retryable: false }; - } - // Network / fetch. - if (msg.includes("aborted") || msg.includes("timeout") || msg.includes("timed out")) { - return { code: "fetch.timeout", kind: "transient", retryable: true }; - } - if (msg.includes("network connection lost") || msg.includes("econnreset") || msg.includes("ehostunreach")) { - return { code: "fetch.error", kind: "transient", retryable: true }; - } - // HTTP status codes embedded in the message (e.g. "status_503", "status=404", - // "status: 502", or a bare " 404 " token). - const statusMatch = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); - if (statusMatch) { - const s = Number(statusMatch[1]); - if (s === 429) - return { code: "rate_limited", kind: "transient", retryable: true }; - if (s >= 500) - return { code: "fetch.http_5xx", kind: "transient", retryable: true }; - if (s >= 400) - return { code: "fetch.http_4xx", kind: "permanent", retryable: false }; - } - // D1 / Vectorize / Workers AI. - if (msg.includes("d1_error") || msg.includes("sqlite_") || msg.includes("database is locked")) { - return { code: "db_error", kind: "transient", retryable: true }; - } - if (msg.includes("vectorize")) - return { code: "vectorize_error", kind: "upstream", retryable: true }; - if (msg.includes("ai.run") || msg.includes("workers ai")) - return { code: "ai_error", kind: "upstream", retryable: true }; - // Parsing. - if (msg.includes("unexpected token") || msg.includes("json")) - return { code: "json_parse_error", kind: "validation", retryable: false }; - if (msg.includes("invalid url") || msg.includes("uri malformed")) - return { code: "validation_failed", kind: "validation", retryable: false }; - // Auth. - if (msg.includes("jwt") || msg.includes("unauthorized") || msg.includes("forbidden")) { - return { code: "unauthorized", kind: "auth", retryable: false }; - } - return null; -} -/** - * Task #72: is this thrown value a benign policy skip (robots.txt / ToS) - * rather than a real fetch failure? Benign skips end a job in the `skipped` - * terminal status (no error_log, no retry), NOT failed/dead_letter. Returns - * the stable `skip_code` + a human reason, or null for everything else (which - * the caller then classifies / retries / dead-letters normally). - */ -export function isBenignSkip(err) { - const cls = err instanceof AppError - ? { code: err.code, kind: err.kind } - : classify(err); - if (!cls || cls.kind !== "skip") - return null; - const skip_code = cls.code === "tos_blocked" ? "tos_blocked" : "robots_disallow"; - const reason = err instanceof Error ? err.message : String(err ?? skip_code); - return { skip_code, reason }; -} -function safeJson(v) { - try { - return JSON.stringify(v); - } - catch { - return String(v); - } -} diff --git a/apps/worker/test-dist-q/personas/repo.js b/apps/worker/test-dist-q/personas/repo.js deleted file mode 100644 index e0170927..00000000 --- a/apps/worker/test-dist-q/personas/repo.js +++ /dev/null @@ -1,403 +0,0 @@ -// Task #46: persona data-access layer. -export const PERSONA_FIELDS = [ - "name", "kind", "status", "thesis", - "hard_filters_json", "size_min", "size_max", "size_bands_json", - "geos_json", "industries_json", - "techs_required_json", "techs_preferred_json", "techs_excluded_json", - "signal_kinds_json", "buyer_titles_json", "buyer_seniority_json", "buyer_departments_json", - "weights_json", "semantic_fit_threshold", "recency_boost", -]; -function parseJsonArr(s) { - if (!s) - return []; - try { - const v = JSON.parse(s); - return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; - } - catch { - return []; - } -} -function parseJsonObj(s) { - if (!s) - return {}; - try { - const v = JSON.parse(s); - return v && typeof v === "object" && !Array.isArray(v) ? v : {}; - } - catch { - return {}; - } -} -export function rowToSpec(row) { - const w = parseJsonObj(row.weights_json); - // Legacy scorer only understands account/buyer. New taxonomy kinds - // fall through to "buyer" for spec purposes (the kind dispatcher in - // services/personas/kinds owns real matching for new kinds; this - // spec is only used by the legacy persona_matches code path). - const legacyKind = row.kind === "account" || row.kind === "account_company" ? "account" : "buyer"; - return { - id: row.id, - kind: legacyKind, - size_min: row.size_min, - size_max: row.size_max, - size_bands: parseJsonArr(row.size_bands_json), - geos: parseJsonArr(row.geos_json), - industries: parseJsonArr(row.industries_json), - techs_required: parseJsonArr(row.techs_required_json), - techs_preferred: parseJsonArr(row.techs_preferred_json), - techs_excluded: parseJsonArr(row.techs_excluded_json), - signal_kinds: parseJsonArr(row.signal_kinds_json), - buyer_titles: parseJsonArr(row.buyer_titles_json), - buyer_seniority: parseJsonArr(row.buyer_seniority_json), - buyer_departments: parseJsonArr(row.buyer_departments_json), - hard_filters: parseJsonObj(row.hard_filters_json), - weights: w, - semantic_fit_threshold: row.semantic_fit_threshold ?? 0.55, - recency_boost: row.recency_boost ?? 0, - }; -} -export async function listPersonas(env, opts) { - const status = opts?.status ?? "active"; - const limit = Math.min(Math.max(1, opts?.limit ?? 200), 500); - const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = ? ORDER BY last_modified DESC LIMIT ?`).bind(status, limit).all(); - return r.results ?? []; -} -export async function getPersona(env, id) { - const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ? AND deleted_at IS NULL`).bind(id).first(); - return r ?? null; -} -export async function getPersonaIncludingDeleted(env, id) { - const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ?`).bind(id).first(); - return r ?? null; -} -export async function insertPersona(env, body, by, idOverride) { - const id = idOverride ?? crypto.randomUUID(); - const now = new Date().toISOString(); - const cols = ["id", "created_by", "created_at", "updated_at", "last_modified", ...PERSONA_FIELDS]; - const binds = [id, by ?? null, now, now, now]; - // Defaults for NOT NULL columns when the caller omits them (e.g. the - // seed loader). Without this, the seed path threw - // `NOT NULL constraint failed: personas.status` and surfaced as a - // db_error on the Personas page. - const defaults = { status: "active", kind: "account" }; - for (const f of PERSONA_FIELDS) { - const v = body[f]; - binds.push(v ?? defaults[f] ?? null); - } - await env.DB.prepare(`INSERT INTO personas (${cols.join(",")}) VALUES (${cols.map(() => "?").join(",")})`).bind(...binds).run(); - await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'created', ?, ?)`) - .bind(crypto.randomUUID(), id, body.name, by ?? null).run(); - const row = await getPersona(env, id); - return row; -} -export async function updatePersona(env, id, patch, by) { - const cur = await getPersona(env, id); - if (!cur) - return null; - const allowed = new Set(PERSONA_FIELDS); - const sets = []; - const binds = []; - const hist = []; - for (const [k, v] of Object.entries(patch)) { - if (!allowed.has(k)) - continue; - sets.push(`${k} = ?`); - binds.push(v); - const before = cur[k]; - if (before !== v) - hist.push({ field: k, old: before, nw: v }); - } - if (!sets.length) - return cur; - const now = new Date().toISOString(); - binds.push(now, now, id); - await env.DB.prepare(`UPDATE personas SET ${sets.join(", ")}, updated_at = ?, last_modified = ? WHERE id = ?`).bind(...binds).run(); - for (const h of hist) { - await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, old_value, new_value, changed_by) VALUES (?, ?, ?, ?, ?, ?)`) - .bind(crypto.randomUUID(), id, h.field, h.old != null ? String(h.old) : null, h.nw != null ? String(h.nw) : null, by ?? null).run(); - } - return await getPersona(env, id); -} -export async function setPersonaEmbeddingMeta(env, id, dim, text) { - const now = new Date().toISOString(); - await env.DB.prepare(`UPDATE personas SET embedding_dim = ?, embedded_at = ?, embedding_text = ?, updated_at = ?, last_modified = ? WHERE id = ?`) - .bind(dim, now, text, now, now, id).run(); -} -export async function setPersonaNotes(env, id, notes) { - const now = new Date().toISOString(); - await env.DB.prepare(`UPDATE personas SET persona_notes = ?, notes_generated_at = ?, updated_at = ? WHERE id = ?`) - .bind(notes, now, now, id).run(); -} -export async function softDeletePersona(env, id, by) { - const now = new Date().toISOString(); - const r = await env.DB.prepare(`UPDATE personas SET deleted_at = ?, status = 'archived', updated_at = ? WHERE id = ? AND deleted_at IS NULL`).bind(now, now, id).run(); - if ((r.meta?.changes ?? 0) > 0) { - await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'archived', ?, ?)`) - .bind(crypto.randomUUID(), id, now, by ?? null).run(); - return true; - } - return false; -} -export async function upsertMatch(env, args) { - const now = new Date().toISOString(); - await env.DB.prepare(`INSERT INTO persona_matches (persona_id, entity_kind, entity_id, fit_score, hard_filter_pass, components_json, explanation, explanation_at, persona_modified_at, entity_modified_at, computed_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ON CONFLICT(persona_id, entity_kind, entity_id) DO UPDATE SET - fit_score = excluded.fit_score, - hard_filter_pass = excluded.hard_filter_pass, - components_json = excluded.components_json, - -- Drop the cached AI explanation when the new score falls below - -- the explanation threshold so we don't keep stale "why this - -- fits" rationale next to a low score. When the new score is - -- still high but no fresh explanation was generated this pass - -- (e.g. budget cap), keep the previous text — explanation_at - -- preserves the original timestamp so callers can detect age. - explanation = CASE - WHEN excluded.fit_score < 50 THEN NULL - WHEN excluded.explanation IS NOT NULL THEN excluded.explanation - ELSE persona_matches.explanation - END, - explanation_at = CASE - WHEN excluded.fit_score < 50 THEN NULL - WHEN excluded.explanation IS NOT NULL THEN excluded.explanation_at - ELSE persona_matches.explanation_at - END, - persona_modified_at = excluded.persona_modified_at, - entity_modified_at = excluded.entity_modified_at, - computed_at = excluded.computed_at`).bind(args.persona_id, args.entity_kind, args.entity_id, args.fit_score, args.hard_filter_pass, JSON.stringify(args.components), args.explanation, args.explanation ? now : null, args.persona_modified_at, args.entity_modified_at, now).run(); -} -export async function listMatches(env, personaId, opts) { - const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); - const offset = Math.max(0, opts.offset ?? 0); - const minScore = Math.max(0, opts.minScore ?? 0); - const kind = opts.kind ?? "account"; - if (kind === "account") { - const r = await env.DB.prepare(`SELECT pm.*, a.name AS entity_name, a.domain AS entity_domain, a.industry AS entity_industry, a.employees AS entity_employees, a.account_score AS entity_account_score - FROM persona_matches pm - JOIN accounts a ON a.id = pm.entity_id - WHERE pm.persona_id = ? AND pm.entity_kind = 'account' AND pm.fit_score >= ? - ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); - return r.results ?? []; - } - const r = await env.DB.prepare(`SELECT pm.*, b.name AS entity_name, b.title AS entity_title, b.seniority AS entity_seniority, b.account_id AS entity_account_id - FROM persona_matches pm - JOIN buyers b ON b.id = pm.entity_id - WHERE pm.persona_id = ? AND pm.entity_kind = 'buyer' AND pm.fit_score >= ? - ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); - return r.results ?? []; -} -export async function countMatches(env, personaId, minScore = 60) { - const r = await env.DB.prepare(`SELECT COUNT(*) AS c FROM persona_matches WHERE persona_id = ? AND fit_score >= ?`).bind(personaId, minScore).first(); - return r?.c ?? 0; -} -export async function deleteMatchesForPersona(env, personaId) { - await env.DB.prepare(`DELETE FROM persona_matches WHERE persona_id = ?`).bind(personaId).run(); -} -export async function listMatchesForEntity(env, entityKind, entityId) { - const r = await env.DB.prepare(`SELECT persona_id, fit_score FROM persona_matches - WHERE entity_kind = ? AND entity_id = ? - ORDER BY fit_score DESC`).bind(entityKind, entityId).all(); - return r.results ?? []; -} -// Task #58: surface persona-fit on the account/buyer detail pages. -// Joins persona_matches with personas so the dashboard can render a -// "Persona fit" panel without a second round-trip per row. Returns -// rows above `minScore` (default 50, matching the explanation cache -// floor in upsertMatch) sorted by score desc. Skips archived/deleted -// personas — matches against an archived persona are stale evidence. -export async function listMatchesForEntityWithDetails(env, entityKind, entityId, opts) { - const minScore = Math.max(0, opts?.minScore ?? 50); - const personaKind = opts?.personaKind ?? entityKind; - const r = await env.DB.prepare(`SELECT pm.persona_id, pm.fit_score, pm.hard_filter_pass, pm.components_json, - pm.explanation, pm.explanation_at, pm.computed_at, - p.name AS persona_name, p.kind AS persona_kind, - p.status AS persona_status, p.thesis AS persona_thesis - FROM persona_matches pm - JOIN personas p ON p.id = pm.persona_id - WHERE pm.entity_kind = ? AND pm.entity_id = ? - AND pm.fit_score >= ? - AND p.deleted_at IS NULL - AND p.status = 'active' - AND p.kind = ? - ORDER BY pm.fit_score DESC`).bind(entityKind, entityId, minScore, personaKind).all(); - return (r.results ?? []).map((row) => ({ - persona_id: row.persona_id, - persona_name: row.persona_name, - persona_kind: row.persona_kind, - persona_status: row.persona_status, - persona_thesis: row.persona_thesis, - fit_score: row.fit_score, - hard_filter_pass: row.hard_filter_pass, - components: row.components_json ? (() => { try { - return JSON.parse(row.components_json); - } - catch { - return null; - } })() : null, - explanation: row.explanation, - explanation_at: row.explanation_at, - computed_at: row.computed_at, - })); -} -// ----- entity fact loaders shared by scorer + workflow -export async function loadAccountFacts(env, accountId) { - const a = await env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id = ?`).bind(accountId).first(); - if (!a) - return null; - const tech = await env.DB.prepare(`SELECT vendor FROM account_tech WHERE account_id = ?`).bind(accountId).all(); - const sigs = await env.DB.prepare(`SELECT kind, weight, confidence, occurred_at FROM signals WHERE account_id = ? ORDER BY occurred_at DESC LIMIT 200`).bind(accountId).all(); - const buyers = await env.DB.prepare(`SELECT id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id = ? LIMIT 50`).bind(accountId).all(); - const facts = { - status: a.status, - domain: a.domain, - hq_country_iso2: a.hq_country_iso2, - size_band: a.size_band, - employees: a.employees, - industry: a.industry, - industries: parseJsonArr(a.industries_json), - funding_stage: a.funding_stage, - techs: (tech.results ?? []).map((t) => t.vendor), - signals: sigs.results ?? [], - buyers: (buyers.results ?? []).map((b) => ({ - account: null, title: b.title, seniority: b.seniority, department: b.department, - is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, - })), - last_modified: a.updated_at, - }; - return { name: a.name, facts, last_modified: a.updated_at }; -} -export async function loadBuyerFacts(env, buyerId) { - const b = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id = ?`).bind(buyerId).first(); - if (!b) - return null; - const acct = await loadAccountFacts(env, b.account_id); - const facts = { - account: acct?.facts ?? null, - title: b.title, seniority: b.seniority, department: b.department, - is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, - }; - return { name: b.name ?? b.title ?? buyerId, facts, last_modified: b.updated_at, account_id: b.account_id }; -} -// Bulk loader for AccountFacts. Issues 4 set-based queries (accounts + -// account_tech + signals + buyers, each with WHERE id IN (?...)) and -// stitches them together. Used by rescorePersonaFull and the -// /preview endpoint to avoid N round-trips over the D1 binding. -export async function loadAccountFactsBulk(env, ids) { - const out = new Map(); - if (!ids.length) - return out; - const ph = ids.map(() => "?").join(","); - const [a, tech, sigs, buyers] = await Promise.all([ - env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id IN (${ph})`).bind(...ids).all(), - env.DB.prepare(`SELECT account_id, vendor FROM account_tech WHERE account_id IN (${ph})`).bind(...ids).all(), - env.DB.prepare(`SELECT account_id, kind, weight, confidence, occurred_at FROM signals WHERE account_id IN (${ph}) ORDER BY occurred_at DESC`).bind(...ids).all(), - env.DB.prepare(`SELECT id, account_id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id IN (${ph})`).bind(...ids).all(), - ]); - const techByAcct = new Map(); - for (const t of tech.results ?? []) { - const arr = techByAcct.get(t.account_id) ?? []; - arr.push(t.vendor); - techByAcct.set(t.account_id, arr); - } - const sigByAcct = new Map(); - for (const s of sigs.results ?? []) { - const arr = sigByAcct.get(s.account_id) ?? []; - if (arr.length < 200) - arr.push({ kind: s.kind, weight: s.weight, confidence: s.confidence, occurred_at: s.occurred_at }); - sigByAcct.set(s.account_id, arr); - } - const buyByAcct = new Map(); - for (const b of buyers.results ?? []) { - const arr = buyByAcct.get(b.account_id) ?? []; - if (arr.length < 50) - arr.push({ account: null, title: b.title, seniority: b.seniority, department: b.department, is_decision_maker: b.is_decision_maker, last_modified: b.updated_at }); - buyByAcct.set(b.account_id, arr); - } - for (const row of a.results ?? []) { - const facts = { - status: row.status, domain: row.domain, hq_country_iso2: row.hq_country_iso2, - size_band: row.size_band, employees: row.employees, industry: row.industry, - industries: parseJsonArr(row.industries_json), funding_stage: row.funding_stage, - techs: techByAcct.get(row.id) ?? [], - signals: sigByAcct.get(row.id) ?? [], - buyers: buyByAcct.get(row.id) ?? [], - last_modified: row.updated_at, - }; - out.set(row.id, { name: row.name, facts, last_modified: row.updated_at }); - } - return out; -} -// Bulk loader for BuyerFacts. Issues 1 query for the buyers + delegates -// to loadAccountFactsBulk for parent accounts (one round-trip via IN). -export async function loadBuyerFactsBulk(env, ids) { - const out = new Map(); - if (!ids.length) - return out; - const ph = ids.map(() => "?").join(","); - const r = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id IN (${ph})`).bind(...ids).all(); - const buyerRows = r.results ?? []; - const acctIds = Array.from(new Set(buyerRows.map((b) => b.account_id))); - const acctFacts = await loadAccountFactsBulk(env, acctIds); - for (const b of buyerRows) { - const acct = acctFacts.get(b.account_id); - const facts = { - account: acct?.facts ?? null, title: b.title, seniority: b.seniority, department: b.department, - is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, - }; - out.set(b.id, { name: b.name ?? b.title ?? b.id, facts, last_modified: b.updated_at, account_id: b.account_id }); - } - return out; -} -// Bulk writeback: recompute max active-persona fit_score for each id -// in one aggregate query, then UPDATE in one statement per kind. Used -// at the end of each rescore batch instead of N per-row writebacks. -export async function bulkWriteBackFit(env, kind, ids) { - if (!ids.length) - return; - const ph = ids.map(() => "?").join(","); - const rows = await env.DB.prepare(`SELECT pm.entity_id AS id, MAX(pm.fit_score) AS m - FROM persona_matches pm - JOIN personas p ON p.id = pm.persona_id - WHERE pm.entity_kind = ? AND pm.entity_id IN (${ph}) - AND p.status = 'active' AND p.deleted_at IS NULL - GROUP BY pm.entity_id`).bind(kind, ...ids).all(); - const maxById = new Map(); - for (const r of rows.results ?? []) - maxById.set(r.id, r.m ?? 0); - // Issue updates as a batch (each binds its own params; D1 batches - // these into one HTTP round-trip via the binding's batch() API). - const stmts = ids.map((id) => { - const m = maxById.get(id) ?? 0; - if (kind === "account") { - return env.DB.prepare(`UPDATE accounts SET fit_score = ?, account_score = ROUND((0.6 * intent_score) + (0.4 * ?), 2) WHERE id = ?`).bind(m, m, id); - } - return env.DB.prepare(`UPDATE buyers SET fit_score = ? WHERE id = ?`).bind(m, id); - }); - await env.DB.batch(stmts); -} -// Materialize a compact "facts" object for the AI explainer. -export function summarizeAccountForExplanation(name, f) { - return { - name, - domain: f.domain, - industry: f.industry, - employees: f.employees, - size_band: f.size_band, - country: f.hq_country_iso2, - funding_stage: f.funding_stage, - top_techs: f.techs.slice(0, 8), - top_signals: f.signals.slice(0, 5).map((s) => ({ kind: s.kind, weight: s.weight, occurred_at: s.occurred_at })), - top_buyer: f.buyers[0] ? { title: f.buyers[0].title, seniority: f.buyers[0].seniority } : null, - }; -} -export function summarizeBuyerForExplanation(name, f) { - return { - name, - title: f.title, - seniority: f.seniority, - department: f.department, - is_decision_maker: !!f.is_decision_maker, - account: f.account ? summarizeAccountForExplanation(name, f.account) : null, - }; -} diff --git a/apps/worker/test-dist-q/personas/score.js b/apps/worker/test-dist-q/personas/score.js deleted file mode 100644 index 4cd5e110..00000000 --- a/apps/worker/test-dist-q/personas/score.js +++ /dev/null @@ -1,281 +0,0 @@ -// Task #46: deterministic persona scoring. -// -// fit_score = clamp(0..100, recency_boost * Σ w_i * c_i) when hard -// filters pass, else 0. Components are each 0..100. semantic_fit comes -// from cosine similarity against the entity's existing embedding (we -// pass it in pre-computed); the rest are computed here from row data. -export const DEFAULT_WEIGHTS_ACCOUNT = { - size: 0.10, - geo: 0.10, - industry: 0.20, - tech: 0.10, - signal: 0.20, - buyer: 0.10, - semantic: 0.20, -}; -export const DEFAULT_WEIGHTS_BUYER = { - size: 0.05, - geo: 0.05, - industry: 0.15, - tech: 0.05, - signal: 0.10, - buyer: 0.45, - semantic: 0.15, -}; -const DAY = 86_400_000; -function lc(s) { return (s ?? "").toLowerCase().trim(); } -function clamp(n, lo, hi) { return Math.max(lo, Math.min(hi, n)); } -function scoreSize(p, emp, band) { - if (p.size_min == null && p.size_max == null && p.size_bands.length === 0) - return 50; - if (band && p.size_bands.length && p.size_bands.includes(band)) - return 100; - if (emp == null) - return 25; - const lo = p.size_min ?? 0; - const hi = p.size_max ?? Number.MAX_SAFE_INTEGER; - if (emp >= lo && emp <= hi) - return 100; - // Soft penalty: 25% per order-of-magnitude away. - const ratio = emp < lo ? lo / Math.max(1, emp) : emp / hi; - const decades = Math.log10(ratio); - return Math.max(0, 100 - Math.round(decades * 60)); -} -function scoreGeo(p, iso) { - if (!p.geos.length) - return 50; - if (!iso) - return 20; - const i = lc(iso); - if (p.geos.includes(i)) - return 100; - // Region bucketing - const REGION = { - emea: ["gb", "ie", "fr", "de", "es", "it", "nl", "be", "se", "no", "fi", "dk", "pt", "pl", "ch", "at"], - apac: ["jp", "sg", "au", "nz", "kr", "in", "hk", "tw", "my", "id", "ph", "th", "vn"], - latam: ["br", "mx", "ar", "cl", "co", "pe", "uy"], - africa: ["za", "ng", "ke", "eg", "ma"], - }; - for (const slug of p.geos) { - const ctry = REGION[slug]; - if (ctry && ctry.includes(i)) - return 80; - if (slug === "global") - return 60; - } - return 0; -} -function scoreIndustry(p, primary, all) { - if (!p.industries.length) - return 50; - const want = new Set(p.industries.map(lc)); - if (primary && want.has(lc(primary))) - return 100; - for (const i of all.map(lc)) - if (want.has(i)) - return 80; - return 0; -} -function scoreTech(p, techs) { - const have = new Set(techs.map(lc)); - if (p.techs_excluded.some((t) => have.has(lc(t)))) - return { score: 0, pass: false }; - if (p.techs_required.length) { - const all = p.techs_required.every((t) => have.has(lc(t))); - if (!all) - return { score: 0, pass: false }; - } - if (!p.techs_preferred.length && !p.techs_required.length) - return { score: 50, pass: true }; - const matched = p.techs_preferred.filter((t) => have.has(lc(t))).length; - const ratio = p.techs_preferred.length ? matched / p.techs_preferred.length : 1; - return { score: Math.round(60 + 40 * ratio), pass: true }; -} -function scoreSignals(p, sigs) { - if (!sigs.length) - return 0; - const want = new Set(p.signal_kinds.map(lc)); - const now = Date.now(); - let totalCredit = 0; - for (const s of sigs) { - const t = Date.parse(s.occurred_at); - const age = Number.isFinite(t) ? Math.max(0, (now - t) / DAY) : 365; - const decay = Math.exp(-age / 30); // half-life-ish - const w = (s.weight ?? 0) * (s.confidence ?? 1); - const kindMul = want.size === 0 ? 0.5 : want.has(lc(s.kind)) ? 1.0 : 0.25; - totalCredit += w * decay * kindMul; - } - return Math.round(100 * (1 - Math.exp(-totalCredit / 15))); -} -function scoreBuyer(p, b) { - let s = 0; - let denom = 0; - if (p.buyer_titles.length) { - denom += 50; - const t = lc(b.title); - if (t && p.buyer_titles.some((x) => t.includes(lc(x)))) - s += 50; - } - if (p.buyer_seniority.length) { - denom += 30; - if (b.seniority && p.buyer_seniority.includes(lc(b.seniority))) - s += 30; - } - if (p.buyer_departments.length) { - denom += 20; - if (b.department && p.buyer_departments.includes(lc(b.department))) - s += 20; - } - if (denom === 0) - return 50; - // Decision-maker bonus - if (b.is_decision_maker) - s = Math.min(denom, s + 5); - return Math.round((s / denom) * 100); -} -function scoreBuyersForAccount(p, buyers) { - if (!buyers.length) - return p.buyer_titles.length || p.buyer_seniority.length ? 0 : 50; - let best = 0; - for (const b of buyers) - best = Math.max(best, scoreBuyer(p, b)); - return best; -} -export function checkHardFilters(p, account, buyer) { - const reasons = []; - const f = p.hard_filters || {}; - const acc = account ?? buyer?.account ?? null; - if (f.require_domain && acc && !acc.domain) { - reasons.push("missing_domain"); - return { pass: false, reasons }; - } - if (Array.isArray(f.statuses_in) && acc && !f.statuses_in.map(lc).includes(lc(acc.status))) { - reasons.push(`status_not_in:${acc.status}`); - return { pass: false, reasons }; - } - if (Array.isArray(f.exclude_country_iso2) && acc && acc.hq_country_iso2 && f.exclude_country_iso2.map(lc).includes(lc(acc.hq_country_iso2))) { - reasons.push(`country_excluded:${acc.hq_country_iso2}`); - return { pass: false, reasons }; - } - if (Array.isArray(f.country_iso2_in) && acc && (!acc.hq_country_iso2 || !f.country_iso2_in.map(lc).includes(lc(acc.hq_country_iso2)))) { - reasons.push("country_not_in"); - return { pass: false, reasons }; - } - if (Array.isArray(f.funding_stage_in) && acc && (!acc.funding_stage || !f.funding_stage_in.map(lc).includes(lc(acc.funding_stage)))) { - reasons.push("funding_stage_not_in"); - return { pass: false, reasons }; - } - if (f.is_decision_maker && buyer && !buyer.is_decision_maker) { - reasons.push("not_decision_maker"); - return { pass: false, reasons }; - } - return { pass: true, reasons }; -} -export function recencyBoost(p, lastModifiedISO) { - // Override wins; otherwise small boost (max 1.15x) for entities updated - // within the last 7 days. Capped 1.0..1.2 per spec. - if (p.recency_boost && p.recency_boost > 0) - return clamp(p.recency_boost, 1.0, 1.2); - if (!lastModifiedISO) - return 1.0; - const t = Date.parse(lastModifiedISO); - if (!Number.isFinite(t)) - return 1.0; - const ageDays = Math.max(0, (Date.now() - t) / DAY); - if (ageDays <= 7) - return 1.15; - if (ageDays <= 30) - return 1.05; - return 1.0; -} -export function scoreEntity(p, ctx) { - const reasons = []; - const hf = checkHardFilters(p, ctx.account, ctx.buyer); - if (!hf.pass) { - return { - fit_score: 0, - components: { - hard_filter_pass: 0, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, - signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, - weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, - reasons: hf.reasons, - }, - }; - } - const acc = ctx.account ?? ctx.buyer?.account ?? null; - const tech = scoreTech(p, acc?.techs ?? []); - if (!tech.pass) { - reasons.push("tech_excluded_or_missing_required"); - return { - fit_score: 0, - components: { - hard_filter_pass: 1, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, - signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, - weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, - reasons, - }, - }; - } - const size = scoreSize(p, acc?.employees ?? null, acc?.size_band ?? null); - const geo = scoreGeo(p, acc?.hq_country_iso2 ?? null); - const industry = scoreIndustry(p, acc?.industry ?? null, acc?.industries ?? []); - const signal = scoreSignals(p, acc?.signals ?? []); - const buyerScore = ctx.buyer ? scoreBuyer(p, ctx.buyer) : scoreBuyersForAccount(p, acc?.buyers ?? []); - const semCos = typeof ctx.semanticCosine === "number" ? ctx.semanticCosine : null; - const semantic = semCos == null - ? 50 - : semCos < (p.semantic_fit_threshold ?? 0.55) ? 0 : Math.round(clamp((semCos - 0.4) / 0.5, 0, 1) * 100); - const w = { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }; - const sumW = (w.size ?? 0) + (w.geo ?? 0) + (w.industry ?? 0) + (w.tech ?? 0) + (w.signal ?? 0) + (w.buyer ?? 0) + (w.semantic ?? 0); - const norm = sumW > 0 ? sumW : 1; - const blended = ((size * (w.size ?? 0)) + - (geo * (w.geo ?? 0)) + - (industry * (w.industry ?? 0)) + - (tech.score * (w.tech ?? 0)) + - (signal * (w.signal ?? 0)) + - (buyerScore * (w.buyer ?? 0)) + - (semantic * (w.semantic ?? 0))) / norm; - const boost = recencyBoost(p, ctx.buyer?.last_modified ?? acc?.last_modified ?? null); - const fit = Math.round(clamp(blended * boost, 0, 100)); - return { - fit_score: fit, - components: { - hard_filter_pass: 1, - size_fit: size, - geo_fit: geo, - industry_fit: industry, - tech_fit: tech.score, - signal_fit: signal, - buyer_fit: buyerScore, - semantic_fit: semantic, - recency_boost: boost, - weights: w, - reasons, - }, - }; -} -export function buildEmbeddingText(p) { - const parts = []; - parts.push(`Persona: ${p.name}`); - if (p.thesis) - parts.push(`Thesis: ${p.thesis}`); - if (p.industries.length) - parts.push(`Industries: ${p.industries.join(", ")}`); - if (p.geos.length) - parts.push(`Geos: ${p.geos.join(", ")}`); - if (p.size_min || p.size_max) - parts.push(`Size: ${p.size_min ?? "?"}–${p.size_max ?? "?"} FTE`); - if (p.techs_required.length) - parts.push(`Required tech: ${p.techs_required.join(", ")}`); - if (p.techs_preferred.length) - parts.push(`Preferred tech: ${p.techs_preferred.join(", ")}`); - if (p.signal_kinds.length) - parts.push(`Watch signals: ${p.signal_kinds.join(", ")}`); - if (p.buyer_titles.length) - parts.push(`Buyer titles: ${p.buyer_titles.join(", ")}`); - if (p.buyer_seniority.length) - parts.push(`Buyer seniority: ${p.buyer_seniority.join(", ")}`); - if (p.buyer_departments.length) - parts.push(`Buyer departments: ${p.buyer_departments.join(", ")}`); - return parts.join(" | "); -} diff --git a/apps/worker/test-dist-q/scraper/firms_upsert.js b/apps/worker/test-dist-q/scraper/firms_upsert.js deleted file mode 100644 index 5dbae1ad..00000000 --- a/apps/worker/test-dist-q/scraper/firms_upsert.js +++ /dev/null @@ -1,267 +0,0 @@ -import { extractDomain } from "./normalize"; -import { syncFirmToEntity } from "../entities/dualwrite"; -const SCALAR_FIELDS = [ - "legal_name", "kind", "website", "logo_url", - "hq_country_iso2", "hq_region", "hq_city", - "thesis", "check_size_min_usd", "check_size_max_usd", "check_size_typical_usd", - "aum_usd", "fund_count", "current_fund_name", "current_fund_size_usd", - "lead_or_co", "portfolio_count", "founded_year", "team_size", - "linkedin_url", "crunchbase_url", "twitter_handle", - "signal_nfx_url", "openvc_url", "pitchbook_url", - "contact_email", "submission_url", -]; -// `source_url` is intentionally excluded from SCALAR_FIELDS — Task #1 -// requires that re-imports from different Folk shares union the -// provenance URLs at the firm row level rather than fill-if-empty. The -// merge path below comma-joins distinct values (mirrors `imported_from`). -const ARRAY_FIELDS = [ - { key: "geo_focus", column: "geo_focus_json" }, - { key: "stages", column: "stages_json" }, - { key: "sectors", column: "sectors_json" }, - { key: "notable_investments", column: "notable_investments_json" }, -]; -export async function upsertFirm(env, candidate, importedFrom, -/** - * Task #1: optional dual-write provenance override. Folk-share imports - * pass `{ source: 'folk_share', sourceKind: 'import' }` so the firm's - * unified-graph facts carry import provenance instead of the default - * `source_kind='scrape'`. - */ -importCtx) { - const name = candidate.name?.trim(); - if (!name) - throw new Error("upsertFirm: candidate.name required"); - const rawDomain = candidate.domain ?? deriveDomain(candidate.website); - const domain = rawDomain ? rawDomain.toLowerCase().trim() : null; - if (domain) - candidate.domain = domain; - // Quality gate: require name + (domain OR website). Without either, - // dedupe is impossible and reruns would create endless duplicates. - if (!domain && !candidate.website) { - throw new Error("upsertFirm: candidate must have domain or website"); - } - const lname = name.toLowerCase(); - // Dedupe lookup. Match on the effective domain (stored.domain coalesced - // with the parsed hostname of stored.website) so a rerun that supplies - // a domain still merges with a row that originally only had a website. - const candidates = await env.DB.prepare("SELECT * FROM firms WHERE lower(name) = ? LIMIT 50").bind(lname).all(); - const rows = candidates.results ?? []; - let existing = null; - for (const r of rows) { - const storedDomain = r.domain ?? deriveDomain(r.website ?? undefined); - if (domain && storedDomain && storedDomain.toLowerCase() === domain) { - existing = r; - break; - } - if (!domain && !storedDomain) { - existing = r; - break; - } - } - const result = existing - ? await mergeInto(env, existing, candidate, importedFrom) - : await insertNew(env, candidate, domain, importedFrom); - // Task #4: dual-write into the unified entity graph (best-effort — - // never block the legacy firm-list importer on a unified-model error). - try { - await syncFirmToEntity(env, { - id: result.firmId, - name: candidate.name, - legal_name: candidate.legal_name ?? null, - website: result.website, - domain: result.domain, - hq_country_iso2: candidate.hq_country_iso2 ?? null, - hq_region: candidate.hq_region ?? null, - hq_city: candidate.hq_city ?? null, - check_size_min_usd: candidate.check_size_min_usd ?? null, - check_size_max_usd: candidate.check_size_max_usd ?? null, - check_size_typical_usd: candidate.check_size_typical_usd ?? null, - thesis: candidate.thesis ?? null, - linkedin_url: candidate.linkedin_url ?? null, - crunchbase_url: candidate.crunchbase_url ?? null, - twitter_handle: candidate.twitter_handle ?? null, - contact_email: candidate.contact_email ?? null, - sectors_json: candidate.sectors ? JSON.stringify(candidate.sectors) : null, - stages_json: candidate.stages ? JSON.stringify(candidate.stages) : null, - geo_focus_json: candidate.geo_focus ? JSON.stringify(candidate.geo_focus) : null, - kind: candidate.kind ?? null, - }, importCtx?.source ?? importedFrom, importCtx?.sourceKind ?? "scrape"); - } - catch (e) { - console.warn("dualwrite syncFirmToEntity failed", result.firmId, e.message); - } - // Role inference now runs centrally inside syncFirmToEntity (the - // unified entity write path), so we no longer call it here. - return result; -} -async function insertNew(env, c, domain, importedFrom) { - const slug = await pickUniqueSlug(env, c.name, domain); - const cols = ["name", "slug", "domain", "imported_from", "last_modified"]; - const vals = [c.name.trim(), slug, domain, importedFrom, new Date().toISOString()]; - for (const f of SCALAR_FIELDS) { - const v = c[f]; - if (v != null && v !== "") { - cols.push(f); - vals.push(v); - } - } - for (const { key, column } of ARRAY_FIELDS) { - const v = c[key]; - if (v && v.length) { - cols.push(column); - vals.push(JSON.stringify(uniqStringArray(v))); - } - } - if (c.source_url) { - cols.push("source_url"); - vals.push(c.source_url); - } - if (c.socials) { - cols.push("socials_json"); - vals.push(JSON.stringify(c.socials)); - } - if (c.notes) { - cols.push("notes"); - vals.push(c.notes); - } - const placeholders = cols.map(() => "?").join(","); - const r = await env.DB.prepare(`INSERT INTO firms (${cols.join(",")}) VALUES (${placeholders})`).bind(...vals).run(); - const firmId = Number(r.meta.last_row_id); - return { firmId, action: "created", website: c.website ?? null, domain }; -} -async function mergeInto(env, existing, c, importedFrom) { - const sets = []; - const binds = []; - // Scalar: only fill missing values; existing non-null values win. - for (const f of SCALAR_FIELDS) { - const newVal = c[f]; - if (newVal == null || newVal === "") - continue; - if (existing[f] == null || existing[f] === "") { - sets.push(`${f} = ?`); - binds.push(newVal); - } - } - // Array fields: set-union of existing JSON array + new entries. - for (const { key, column } of ARRAY_FIELDS) { - const incoming = c[key]; - if (!incoming || !incoming.length) - continue; - const existingArr = parseJsonArray(existing[column]); - const merged = uniqStringArray([...existingArr, ...incoming]); - if (merged.length !== existingArr.length) { - sets.push(`${column} = ?`); - binds.push(JSON.stringify(merged)); - } - } - if (c.socials) { - const existingSocials = parseJsonObject(existing.socials_json); - const merged = { ...existingSocials, ...c.socials }; - if (Object.keys(merged).length !== Object.keys(existingSocials).length) { - sets.push("socials_json = ?"); - binds.push(JSON.stringify(merged)); - } - } - if (c.notes && !existing.notes) { - sets.push("notes = ?"); - binds.push(c.notes); - } - // Track every distinct origin. - const importedFromExisting = existing.imported_from ?? ""; - if (!importedFromExisting.split(",").includes(importedFrom)) { - sets.push("imported_from = ?"); - binds.push(importedFromExisting ? `${importedFromExisting},${importedFrom}` : importedFrom); - } - // Task #1: union source_url across re-imports. Folk shares (Top-300, - // FR VCs, etc.) each have their own share URL; re-importing the same - // firm from a second share must preserve evidence of both shares - // rather than fill-if-empty (which would silently drop the second - // URL). Mirrors the imported_from comma-join pattern above; the - // unified graph still gets one channel/fact per share via dualwrite. - const newSourceUrl = c.source_url; - if (typeof newSourceUrl === "string" && newSourceUrl) { - const existingSourceUrl = existing.source_url ?? ""; - const parts = existingSourceUrl ? existingSourceUrl.split(",").map((s) => s.trim()).filter(Boolean) : []; - if (!parts.includes(newSourceUrl)) { - parts.push(newSourceUrl); - sets.push("source_url = ?"); - binds.push(parts.join(",")); - } - } - // Always bump last_modified on every dedupe hit — even when no field - // deltas applied — so reruns leave a verifiable timestamp trail. - const action = sets.length ? "updated" : "unchanged"; - sets.push("last_modified = ?"); - binds.push(new Date().toISOString()); - binds.push(existing.id); - await env.DB.prepare(`UPDATE firms SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); - const persistedWebsite = existing.website ?? c.website ?? null; - const persistedDomain = existing.domain ?? deriveDomain(persistedWebsite); - return { firmId: existing.id, action, website: persistedWebsite, domain: persistedDomain }; -} -function deriveDomain(website) { - if (!website) - return null; - return extractDomain(website) || null; -} -function slugify(s) { - return s.toLowerCase().normalize("NFKD").replace(/[^\w]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80); -} -async function pickUniqueSlug(env, name, domain) { - const base = slugify(name) || "firm"; - let candidate = base; - let row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); - if (!row) - return candidate; - if (domain) { - candidate = `${base}-${slugify(domain)}`; - row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); - if (!row) - return candidate; - } - for (let i = 2; i < 100; i++) { - const c = `${base}-${i}`; - row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(c).first(); - if (!row) - return c; - } - // Last resort: random suffix. - return `${base}-${Math.random().toString(36).slice(2, 8)}`; -} -function parseJsonArray(raw) { - if (!raw) - return []; - try { - const v = JSON.parse(raw); - return Array.isArray(v) ? v.map((x) => String(x)) : []; - } - catch { - return []; - } -} -function parseJsonObject(raw) { - if (!raw) - return {}; - try { - const v = JSON.parse(raw); - return v && typeof v === "object" && !Array.isArray(v) ? v : {}; - } - catch { - return {}; - } -} -function uniqStringArray(arr) { - const seen = new Set(); - const out = []; - for (const s of arr) { - const k = String(s).trim(); - if (!k) - continue; - const lk = k.toLowerCase(); - if (seen.has(lk)) - continue; - seen.add(lk); - out.push(k); - } - return out; -} diff --git a/apps/worker/test-dist-q/scraper/normalize.js b/apps/worker/test-dist-q/scraper/normalize.js deleted file mode 100644 index 5f9c29b0..00000000 --- a/apps/worker/test-dist-q/scraper/normalize.js +++ /dev/null @@ -1,104 +0,0 @@ -// Normalization helpers used by the parsers and dedupe key generators. -const EMAIL_RE = /^[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}$/i; -const PHONE_DIGITS_RE = /[^\d+]/g; -export function normalizeEmail(raw) { - if (!raw) - return null; - const s = raw.trim().toLowerCase(); - if (!EMAIL_RE.test(s)) - return null; - return s; -} -/** - * Email key for dedupe purposes only — strips +tags and lowercases the domain. - * The displayed email uses normalizeEmail (preserves the +tag). - */ -export function emailDedupeKey(email) { - const e = normalizeEmail(email); - if (!e) - return null; - const [localPart, domain] = e.split("@"); - const stripped = localPart.split("+")[0]; - return `${stripped}@${domain.toLowerCase()}`; -} -/** - * Best-effort E.164 normalization. We only accept numbers that already include - * a leading + or are clearly internationalizable (10–15 digits). Otherwise null. - */ -export function normalizePhoneE164(raw) { - if (!raw) - return null; - const cleaned = raw.replace(PHONE_DIGITS_RE, ""); - if (!cleaned) - return null; - if (cleaned.startsWith("+")) { - const digits = cleaned.slice(1); - if (digits.length < 7 || digits.length > 15) - return null; - return `+${digits}`; - } - // No country code → don't guess; return null so we don't pollute dedupe keys. - if (cleaned.length === 10 || cleaned.length === 11) { - // Common case: US/Canada 10 digits or 1+10 digits. - const d = cleaned.length === 10 ? `1${cleaned}` : cleaned; - return `+${d}`; - } - return null; -} -export function canonicalizeLinkedinUrl(url) { - if (!url) - return null; - try { - const u = new URL(url); - const host = u.hostname.toLowerCase().replace(/^www\./, ""); - if (!host.endsWith("linkedin.com")) - return null; - // Trailing slash and query strip - const path = u.pathname.replace(/\/+$/, "").toLowerCase(); - if (!/^\/(in|company|school)\/[^\/]+/.test(path)) - return null; - return `https://www.linkedin.com${path}`; - } - catch { - return null; - } -} -export function countryNameToIso2(name) { - if (!name) - return null; - const s = name.trim().toLowerCase(); - // Tiny seed map; the full mapping lives with the taxonomies task. - const seed = { - usa: "US", - "united states": "US", - "united states of america": "US", - canada: "CA", - uk: "GB", - "united kingdom": "GB", - england: "GB", - france: "FR", - germany: "DE", - spain: "ES", - italy: "IT", - netherlands: "NL", - switzerland: "CH", - israel: "IL", - india: "IN", - china: "CN", - japan: "JP", - singapore: "SG", - australia: "AU", - brazil: "BR", - }; - if (s.length === 2) - return s.toUpperCase(); - return seed[s] ?? null; -} -export function extractDomain(url) { - try { - return new URL(url).hostname.toLowerCase().replace(/^www\./, ""); - } - catch { - return ""; - } -} diff --git a/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js b/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js deleted file mode 100644 index cb0ff5c3..00000000 --- a/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js +++ /dev/null @@ -1 +0,0 @@ -export {}; diff --git a/apps/worker/test-dist-q/scraper/rateLimit.js b/apps/worker/test-dist-q/scraper/rateLimit.js deleted file mode 100644 index ebd48fb8..00000000 --- a/apps/worker/test-dist-q/scraper/rateLimit.js +++ /dev/null @@ -1,53 +0,0 @@ -// Per-host + global AI rate limiting (Task #25 step 6). -// -// Prefers the Cloudflare Rate Limiter binding (`RL_HOST`/`RL_AI`) when -// configured. Falls back to a KV-backed leaky-bucket counter on -// SCRAPE_CACHE so the worker keeps pacing itself even when the binding is -// missing in dev or before the namespace is provisioned. -const KV_PREFIX = "rl:"; -const HOST_LIMIT_PER_MIN = 60; -const AI_LIMIT_PER_MIN = 600; -const WINDOW_MS = 60_000; -async function kvLeakyBucket(kv, key, limit) { - if (!kv) - return true; - const raw = await kv.get(key); - const now = Date.now(); - let bucket = raw ? safeParse(raw) : { count: 0, window_start: now }; - if (now - bucket.window_start > WINDOW_MS) - bucket = { count: 0, window_start: now }; - if (bucket.count >= limit) - return false; - bucket.count += 1; - await kv.put(key, JSON.stringify(bucket), { expirationTtl: 120 }); - return true; -} -function safeParse(raw) { - try { - const v = JSON.parse(raw); - if (typeof v?.count === "number" && typeof v?.window_start === "number") - return v; - } - catch { /* swallow */ } - return { count: 0, window_start: Date.now() }; -} -export async function limitHost(env, host) { - if (env.RL_HOST) { - try { - const r = await env.RL_HOST.limit({ key: host }); - return r.success; - } - catch { /* fall through to KV */ } - } - return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}host:${host}`, HOST_LIMIT_PER_MIN); -} -export async function limitAi(env) { - if (env.RL_AI) { - try { - const r = await env.RL_AI.limit({ key: "global" }); - return r.success; - } - catch { /* fall through */ } - } - return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}ai:global`, AI_LIMIT_PER_MIN); -} diff --git a/apps/worker/test-dist-q/services/personaMatchTrigger.js b/apps/worker/test-dist-q/services/personaMatchTrigger.js deleted file mode 100644 index 16f48214..00000000 --- a/apps/worker/test-dist-q/services/personaMatchTrigger.js +++ /dev/null @@ -1,59 +0,0 @@ -// Task #8: per-entity persona-match refresh trigger. -// -// Called from entity write paths (insertFact, addCareerEntry) so a new -// job or relocation flows into persona candidate rankings within -// minutes, not at the next nightly cron. Debounced via KV so a burst -// of fact writes for the same entity only triggers one re-match. -// Predicates that materially affect a person-entity's persona score. -// Other predicates (donations, family ties, lifestyle, etc.) are -// ignored here so we don't dispatch on unrelated edits. -const RELEVANT_PREDICATES = new Set([ - "person.career", "person.title", "title", - "person.seniority", "person.department", - "person.location.country", "person.location.city", - "location.country", "location.city", - "employer", "person.employer", - // `employees` is the spelling that actually gets written; without it a - // fresh headcount fact never re-scored the entity it belongs to. - "employees", - "org.headcount", "org.employees", - "company.employees", "company.headcount", - "org.sector", "sector", - "org.stage", "stage", -]); -export function isRelevantPredicate(predicate) { - if (!predicate) - return false; - return RELEVANT_PREDICATES.has(predicate); -} -const DEBOUNCE_SECONDS = 300; // 5 minutes -export async function triggerEntityMatchRefresh(env, entityId) { - if (!entityId) - return; - // KV debounce — first write wins per 5min window. - try { - const kvKey = `pem:trigger:${entityId}`; - if (env.SESSIONS) { - const existing = await env.SESSIONS.get(kvKey); - if (existing) - return; - await env.SESSIONS.put(kvKey, "1", { expirationTtl: DEBOUNCE_SECONDS }); - } - } - catch (e) { - console.warn("triggerEntityMatchRefresh debounce check failed", entityId, e.message); - // Fall through — better to dispatch than miss the trigger. - } - // Dispatch the per-entity workflow; inline fallback runs the service. - try { - if (env.WF_PERSONA_MATCH_ENTITY) { - await env.WF_PERSONA_MATCH_ENTITY.create({ params: { entityId } }); - return; - } - const { scoreEntityAcrossPersonas } = await import("./personaMatching.js"); - await scoreEntityAcrossPersonas(env, entityId); - } - catch (e) { - console.warn("triggerEntityMatchRefresh dispatch failed", entityId, e.message); - } -} diff --git a/apps/worker/test-dist-q/services/personaMatching.js b/apps/worker/test-dist-q/services/personaMatching.js deleted file mode 100644 index c2e893c1..00000000 --- a/apps/worker/test-dist-q/services/personaMatching.js +++ /dev/null @@ -1,484 +0,0 @@ -// Task #8: Real persona matching algorithm. -// -// Deterministic weighted scoring engine that ranks unified `u_entities` -// (person entities) against a persona. Each entity gets a score in -// [0,1] plus a transparent per-component breakdown so the dashboard -// can explain *why* an entity matched. -// -// Pure scoring primitives live in personaMatchingScorers.ts (no Env -// imports — unit-testable). This module orchestrates the D1 loads, -// the title embedding, and the upsert. -import { aiEmbed } from "../ai/extract"; -import { assertBudget } from "../ai/budget"; -import { getPersona } from "../personas/repo"; -import { DEFAULT_WEIGHTS, MODEL_VERSION, cosine, aggregate, buildRationale, extractTargets, scoreSeniority, scoreFunction, scoreIndustry, scoreCompanySize, scoreStage, scoreGeo, } from "./personaMatchingScorers"; -export { DEFAULT_WEIGHTS, MODEL_VERSION, extractTargets }; -// Task #3: structural-only fallback used when a kind plugin returns -// null (e.g. fund/company targets that have no per-entity scoring -// pipeline). Builds a properly-typed ComponentMap with zeroed -// components so downstream consumers (rationale builder, persistence) -// don't have to special-case the structural row. Replaces an earlier -// `as unknown as MatchResult` cast that bypassed the type system. -export function buildStructuralFallback(reason) { - const components = {}; - for (const key of Object.keys(DEFAULT_WEIGHTS)) { - components[key] = { value: 0, weight: 0, reason: "n/a (structural fallback)" }; - } - return { score: 0.5, components, rationale: reason }; -} -// --------------------------------------------------------------------------- -// Entity loader. -// --------------------------------------------------------------------------- -async function loadEmployerFacts(env, employerId) { - const sum = await env.DB.prepare(`SELECT display_name, country_iso2, sectors_csv, stages_csv FROM entity_summary WHERE entity_id = ?`).bind(employerId).first(); - // `employees` is first because it is the only one of these that anything - // writes: it is the predicate the registry declares - // (entities/profile-predicates.ts) and the one secEdgar/persist.ts and the - // account dual-write emit. The four `org.*` / `company.*` spellings below - // were the entire list, and no writer has ever produced one — so - // `employees` came back null for every entity and scoreCompanySize - // returned its "company size unknown" zero every time. With a weight of - // 0.10 that put a hard ceiling of 0.90 on every persona match, and made - // "company size unknown" a permanent line in the rationale the dashboard - // shows to explain why someone matched. The unused spellings are kept so a - // future writer picking one still resolves. - const hc = await env.DB.prepare(`SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('employees','org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(employerId).first(); - if (!sum && !hc) - return null; - return { - name: sum?.display_name ?? null, - country: sum?.country_iso2 ?? null, - sectors: sum?.sectors_csv ? sum.sectors_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], - stages: sum?.stages_csv ? sum.stages_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], - employees: hc?.value_number != null ? Math.round(hc.value_number) : null, - }; -} -async function loadEntityCoords(env, entityId) { - const r = await env.DB.prepare(`SELECT predicate, value_number FROM facts - WHERE entity_id = ? AND is_current = 1 - AND predicate IN ('person.location.lat','person.location.lng','geo.lat','geo.lng','location.lat','location.lng') - AND value_number IS NOT NULL`).bind(entityId).all(); - let lat = null; - let lng = null; - for (const row of r.results ?? []) { - if (lat == null && (row.predicate.endsWith(".lat") || row.predicate === "geo.lat")) - lat = row.value_number; - if (lng == null && (row.predicate.endsWith(".lng") || row.predicate === "geo.lng")) - lng = row.value_number; - } - return { lat, lng }; -} -export async function loadPersonEntity(env, entityId) { - const ent = await env.DB.prepare(`SELECT id, display_name, kind, status FROM u_entities WHERE id = ?`).bind(entityId).first(); - if (!ent) - return null; - if (ent.kind !== "person") - return null; - if (ent.status === "merged" || ent.status === "soft_deleted") - return null; - const sum = await env.DB.prepare(`SELECT country_iso2, region FROM entity_summary WHERE entity_id = ?`).bind(entityId).first(); - const career = await env.DB.prepare(`SELECT role_title, seniority, department, organization_entity_id, organization_name - FROM career_history - WHERE entity_id = ? - ORDER BY is_current DESC, COALESCE(ended_at, '9999') DESC, started_at DESC - LIMIT 1`).bind(entityId).first(); - let title = career?.role_title ?? null; - if (!title) { - const tf = await env.DB.prepare(`SELECT value_text FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('person.title','title') AND value_text IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(entityId).first(); - title = tf?.value_text ?? null; - } - let employer = null; - if (career?.organization_entity_id) { - try { - employer = await loadEmployerFacts(env, career.organization_entity_id); - } - catch { /* ignore */ } - } - const coords = await loadEntityCoords(env, entityId).catch(() => ({ lat: null, lng: null })); - return { - id: ent.id, - display_name: ent.display_name, - country_iso2: sum?.country_iso2 ?? null, - region: sum?.region ?? null, - lat: coords.lat, - lng: coords.lng, - title, - seniority: career?.seniority ?? null, - department: career?.department ?? null, - employer_entity_id: career?.organization_entity_id ?? null, - employer_name: employer?.name ?? career?.organization_name ?? null, - employer_country: employer?.country ?? null, - employer_sectors: employer?.sectors ?? [], - employer_stages: employer?.stages ?? [], - employer_employees: employer?.employees ?? null, - }; -} -// --------------------------------------------------------------------------- -// Title similarity (only DB/Env-touching scorer). -// -// Embeddings are cached in persona_title_embeddings + entity_title_embeddings -// keyed by content_hash so the hot path becomes a D1 lookup instead of an -// AI.embed call. AI.embed only fires on cache miss (new persona, new entity, -// or title text change). This mirrors the Vectorize precompute/reuse -// pattern from Task #7 personas while staying on D1. -// --------------------------------------------------------------------------- -async function sha256Hex(input) { - const buf = new TextEncoder().encode(input); - const digest = await crypto.subtle.digest("SHA-256", buf); - return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); -} -async function ensureTitleCacheTables(env) { - try { - await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_title_embeddings (persona_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); - await env.DB.prepare(`CREATE TABLE IF NOT EXISTS entity_title_embeddings (entity_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); - } - catch { /* best-effort */ } -} -async function getOrEmbedTitle(env, scope, id, text) { - const hash = await sha256Hex(text); - const table = scope === "persona" ? "persona_title_embeddings" : "entity_title_embeddings"; - const idCol = scope === "persona" ? "persona_id" : "entity_id"; - try { - const row = await env.DB.prepare(`SELECT vector_json FROM ${table} WHERE ${idCol} = ? AND content_hash = ?`).bind(id, hash).first(); - if (row?.vector_json) { - try { - const v = JSON.parse(row.vector_json); - if (Array.isArray(v) && v.length) - return v; - } - catch { /* fall through to re-embed */ } - } - } - catch { - // Table missing — create it once and continue with embedding path. - await ensureTitleCacheTables(env); - } - if (!env.AI) - return null; - const vec = await aiEmbed(env, text); - if (vec && vec.length) { - try { - await env.DB.prepare(`INSERT INTO ${table} (${idCol}, content_hash, vector_json, updated_at) VALUES (?, ?, ?, datetime('now')) - ON CONFLICT(${idCol}) DO UPDATE SET content_hash=excluded.content_hash, vector_json=excluded.vector_json, updated_at=excluded.updated_at`).bind(id, hash, JSON.stringify(vec)).run(); - } - catch (e) { - console.warn("title embedding cache write failed", scope, id, e.message); - } - } - return vec; -} -async function titleSimilarity(env, personaId, personaText, entityId, entityTitle) { - if (!entityTitle || !personaText) { - return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "missing title" }; - } - if (!env.AI) { - const ov = scoreFunction(entityTitle, [personaText]); - return { value: ov.value, weight: DEFAULT_WEIGHTS.title_sim, reason: `title token overlap (no embed): ${ov.reason}` }; - } - try { - const [pv, ev] = await Promise.all([ - getOrEmbedTitle(env, "persona", personaId, personaText), - getOrEmbedTitle(env, "entity", entityId, entityTitle), - ]); - if (!pv || !ev) { - return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "embedding unavailable" }; - } - const c = cosine(pv, ev); - return { value: c, weight: DEFAULT_WEIGHTS.title_sim, reason: `title cosine ${c.toFixed(3)} ("${entityTitle}")`, data: { cosine: c } }; - } - catch (e) { - return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: `embed error: ${e.message.slice(0, 60)}` }; - } -} -// --------------------------------------------------------------------------- -// Public API. -// --------------------------------------------------------------------------- -export async function scoreEntityForPersona(env, persona, entity) { - const targets = extractTargets(persona); - const [title_sim, seniority, fnc, industry, company_size, stage, geo] = await Promise.all([ - titleSimilarity(env, persona.id, targets.title_text || targets.titles.join(", "), entity.id, entity.title), - Promise.resolve(scoreSeniority(entity.seniority, targets.seniority)), - Promise.resolve(scoreFunction(entity.department, targets.functions)), - Promise.resolve(scoreIndustry(entity.employer_sectors, targets.industries)), - Promise.resolve(scoreCompanySize(entity.employer_employees, targets.size_min, targets.size_max)), - Promise.resolve(scoreStage(entity.employer_stages, targets.stages)), - Promise.resolve(scoreGeo({ - entityIso2: entity.country_iso2 ?? entity.employer_country, - entityLat: entity.lat, entityLng: entity.lng, - targets: targets.geos, - centerLat: targets.geo_center_lat, centerLng: targets.geo_center_lng, - radiusKm: targets.geo_radius_km, - })), - ]); - const components = { title_sim, seniority, function: fnc, industry, company_size, stage, geo }; - const score = aggregate(components); - const rationale = buildRationale(persona.name, entity.display_name, entity.employer_name, components, score); - return { score, components, rationale }; -} -export async function scoreEntity(env, personaId, entityId) { - const persona = await getPersona(env, personaId); - if (!persona) - return null; - const entity = await loadPersonEntity(env, entityId); - if (!entity) - return null; - // Task #2 budget gate: refuse if AI cap reached (title_sim uses AI.embed). - const b = await assertBudget(env, "ai"); - if (!b.ok) { - await recordMatchJob(env, "score_entity", "halted", { personaId, entityId, reason: b.reason }); - return null; - } - const result = await scoreEntityForPersona(env, persona, entity); - await upsertMatch(env, personaId, entityId, result); - return result; -} -// Task #8: durable job/error log so SLO violations are visible. Created -// on demand by triggering migrations; CREATE TABLE IF NOT EXISTS guards -// against pre-migration calls. -async function ensureJobsTable(env) { - try { - await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_match_jobs ( - id TEXT PRIMARY KEY, - kind TEXT NOT NULL, - status TEXT NOT NULL, - persona_id TEXT, - entity_id TEXT, - details_json TEXT, - created_at TEXT NOT NULL DEFAULT (datetime('now')) - )`).run(); - } - catch { /* best-effort */ } -} -export async function recordMatchJob(env, kind, status, details) { - await ensureJobsTable(env); - try { - await env.DB.prepare(`INSERT INTO persona_match_jobs (id, kind, status, persona_id, entity_id, details_json) - VALUES (?, ?, ?, ?, ?, ?)`).bind(crypto.randomUUID(), kind, status, details.personaId ?? null, details.entityId ?? null, JSON.stringify(details)).run(); - } - catch (e) { - console.warn("recordMatchJob failed", kind, status, e.message); - } -} -async function isCancelled(env, jobId) { - if (!jobId) - return false; - try { - const r = await env.DB.prepare("SELECT status FROM jobs WHERE id = ?").bind(jobId).first(); - return r?.status === "cancelled" || r?.status === "timed_out"; - } - catch { - return false; - } -} -export async function upsertMatch(env, personaId, entityId, result, opts = {}) { - const source = opts.source ?? "auto"; - const evidence = JSON.stringify({ - components: Object.fromEntries(Object.keys(result.components).map((k) => [k, { - value: Number(result.components[k].value.toFixed(4)), - weight: result.components[k].weight, - reason: result.components[k].reason, - }])), - rationale: result.rationale, - weights: DEFAULT_WEIGHTS, - version: MODEL_VERSION, - }); - if (source === "auto") { - await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) - VALUES (?, ?, ?, ?, 'auto', datetime('now'), ?) - ON CONFLICT(persona_id, entity_id) DO UPDATE SET - score = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.score ELSE excluded.score END, - match_evidence_json = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.match_evidence_json ELSE excluded.match_evidence_json END, - last_scored_at = excluded.last_scored_at, - model_version = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.model_version ELSE excluded.model_version END`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); - return; - } - await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) - VALUES (?, ?, ?, ?, 'manual', datetime('now'), ?) - ON CONFLICT(persona_id, entity_id) DO UPDATE SET - score = excluded.score, - match_evidence_json = excluded.match_evidence_json, - source = 'manual', - last_scored_at = excluded.last_scored_at, - model_version = excluded.model_version`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); -} -export async function scoreEntityAcrossPersonas(env, entityId, opts = {}) { - const entity = await loadPersonEntity(env, entityId); - if (!entity) - return { scored: 0, errors: 0, halted: false }; - const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = 'active'`).all(); - let scored = 0; - let errors = 0; - let halted = false; - for (const p of r.results ?? []) { - // Task #2: budget + cancellation enforcement per item. - const b = await assertBudget(env, "ai"); - if (!b.ok) { - halted = true; - await recordMatchJob(env, "score_across_personas", "halted", { entityId, scored, errors, reason: b.reason }); - break; - } - if (await isCancelled(env, opts.jobId ?? null)) { - halted = true; - await recordMatchJob(env, "score_across_personas", "cancelled", { entityId, scored, errors }); - break; - } - try { - const res = await scoreEntityForPersona(env, p, entity); - await upsertMatch(env, p.id, entityId, res); - scored += 1; - } - catch (e) { - errors += 1; - console.warn("scoreEntityAcrossPersonas item failed", p.id, entityId, e.message); - } - } - return { scored, errors, halted }; -} -export async function scoreBatch(env, personaId, opts = {}) { - const persona = await getPersona(env, personaId); - if (!persona) - return { scored: 0, errors: 0, pages: 0, halted: false }; - const batchSize = Math.min(Math.max(1, opts.batchSize ?? 100), 500); - // maxEntities = null (default) means "process every active person - // entity" — the task requires create/edit dispatch covers all - // entities, not a hardcoded cap. Operators can pass a number when - // they want to bound a manual run. - const maxEntities = opts.maxEntities ?? null; - let offset = 0; - let scored = 0; - let errors = 0; - let pages = 0; - let halted = false; - for (;;) { - if (maxEntities != null && scored + errors >= maxEntities) - break; - // Task #2: budget + cancellation check per page (cheap, bounded). - const b = await assertBudget(env, "ai"); - if (!b.ok) { - halted = true; - await recordMatchJob(env, "score_batch", "halted", { personaId, scored, errors, pages, reason: b.reason }); - break; - } - if (await isCancelled(env, opts.jobId ?? null)) { - halted = true; - await recordMatchJob(env, "score_batch", "cancelled", { personaId, scored, errors, pages }); - break; - } - // Task #3: dispatch through the kind plugin so each persona kind - // selects its own candidate pool (e.g. investor_person filters - // entity_roles.role IN ('investor','vc','gp','partner_at_firm')). - // Note: explicit .js extension here is required by tsconfig.test.json's - // NodeNext moduleResolution. The wrangler build / typecheck doesn't - // care; this is purely to unblock `pnpm test`. - const { getPluginFor } = await import("./personas/kinds/index.js"); - const plugin = getPluginFor(persona.kind); - const filter = plugin.defaultEntityFilter(persona, { limit: batchSize, offset }); - const r = await env.DB.prepare(filter.sql).bind(...filter.binds).all(); - const ids = (r.results ?? []).map((x) => x.id); - if (!ids.length) - break; - pages += 1; - for (const id of ids) { - try { - // Task #3: delegate to the kind plugin so bespoke matchers - // (investor_firm structural, venture_partner subtype, etc.) - // get the chance to override scoring. The generic plugin's - // scoreEntity returns the person-graph score for person - // targets and null for fund/company targets — when null, we - // persist a deterministic structural-match row at score 50 - // so non-person kinds still surface candidates in the UI. - let res = await plugin.scoreEntity(env, persona, id); - if (!res) - res = buildStructuralFallback(plugin.explainMatch(id)); - await upsertMatch(env, personaId, id, res); - scored += 1; - } - catch (e) { - errors += 1; - console.warn("scoreBatch item failed", personaId, id, e.message); - } - } - if (ids.length < batchSize) - break; - offset += batchSize; - } - if (errors > 0 && !halted) { - await recordMatchJob(env, "score_batch", "ok", { personaId, scored, errors, pages }); - } - return { scored, errors, pages, halted }; -} -export async function refreshStaleMatches(env, opts = {}) { - const staleDays = Math.max(1, opts.staleDays ?? 30); - const limit = Math.min(Math.max(1, opts.limit ?? 500), 5000); - const r = await env.DB.prepare(`SELECT persona_id, entity_id FROM persona_entity_matches - WHERE source = 'auto' AND datetime(last_scored_at) < datetime('now', ?) - ORDER BY last_scored_at ASC LIMIT ?`).bind(`-${staleDays} days`, limit).all(); - let refreshed = 0; - let errors = 0; - let halted = false; - for (const row of r.results ?? []) { - // Task #2: per-item budget + cancellation gate. - const b = await assertBudget(env, "ai"); - if (!b.ok) { - halted = true; - await recordMatchJob(env, "refresh_stale", "halted", { refreshed, errors, reason: b.reason }); - break; - } - if (await isCancelled(env, opts.jobId ?? null)) { - halted = true; - await recordMatchJob(env, "refresh_stale", "cancelled", { refreshed, errors }); - break; - } - try { - const res = await scoreEntity(env, row.persona_id, row.entity_id); - if (res) - refreshed += 1; - } - catch (e) { - errors += 1; - console.warn("refreshStaleMatches item failed", row.persona_id, row.entity_id, e.message); - } - } - return { refreshed, errors, halted }; -} -export async function listCandidates(env, personaId, opts) { - const minScore = Math.max(0, Math.min(1, opts.minScore ?? 0)); - const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); - const offset = Math.max(0, opts.offset ?? 0); - const r = await env.DB.prepare(`SELECT pem.persona_id, pem.entity_id, pem.score, pem.source, pem.last_scored_at, pem.model_version, pem.match_evidence_json, - ue.display_name AS entity_name, ue.primary_domain AS entity_domain, - es.country_iso2 AS entity_country - FROM persona_entity_matches pem - JOIN u_entities ue ON ue.id = pem.entity_id - LEFT JOIN entity_summary es ON es.entity_id = pem.entity_id - WHERE pem.persona_id = ? AND pem.score >= ? - ORDER BY pem.score DESC, pem.last_scored_at DESC - LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); - return (r.results ?? []).map((row) => { - let components = {}; - let rationale = ""; - if (row.match_evidence_json) { - try { - const j = JSON.parse(row.match_evidence_json); - if (j.components) - components = j.components; - if (typeof j.rationale === "string") - rationale = j.rationale; - } - catch { /* ignore */ } - } - return { - persona_id: row.persona_id, - entity_id: row.entity_id, - score: row.score, - source: row.source, - last_scored_at: row.last_scored_at, - model_version: row.model_version, - components, - rationale, - entity_name: row.entity_name, - entity_domain: row.entity_domain, - entity_country: row.entity_country, - }; - }); -} diff --git a/apps/worker/test-dist-q/services/personaMatchingScorers.js b/apps/worker/test-dist-q/services/personaMatchingScorers.js deleted file mode 100644 index 0d68be9d..00000000 --- a/apps/worker/test-dist-q/services/personaMatchingScorers.js +++ /dev/null @@ -1,366 +0,0 @@ -// Task #8: Pure scoring primitives for the persona ↔ entity matcher. -// -// This module is intentionally free of D1 / Env / AI imports so it can -// be unit-tested in node:test (see test/personaMatching.test.mjs). The -// orchestrator (services/personaMatching.ts) wires these primitives to -// the database, embeddings, and workflow dispatch. -export const MODEL_VERSION = "v1"; -export const DEFAULT_WEIGHTS = { - title_sim: 0.25, - seniority: 0.15, - function: 0.15, - industry: 0.15, - company_size: 0.10, - stage: 0.10, - geo: 0.10, -}; -const SENIORITY_LADDER = [ - "ic", "analyst", "associate", "manager", "principal", - "director", "vp", "svp", "cxo", "founder", "partner", -]; -const SENIORITY_INDEX = Object.fromEntries(SENIORITY_LADDER.map((s, i) => [s, i])); -const STAGE_LADDER = [ - "pre_seed", "seed", "series_a", "series_b", "series_c", - "series_d", "growth", "late", "public", -]; -const STAGE_INDEX = Object.fromEntries(STAGE_LADDER.map((s, i) => [s, i])); -const INDUSTRY_PARENTS = { - fintech: ["finance"], insurtech: ["finance"], wealthtech: ["finance"], - proptech: ["realestate"], regtech: ["finance", "compliance"], - edtech: ["education"], healthtech: ["healthcare"], - biotech: ["healthcare", "lifesciences"], medtech: ["healthcare"], - cleantech: ["energy"], climatetech: ["energy"], - saas: ["software"], devtools: ["software"], paas: ["software"], - martech: ["marketing"], adtech: ["marketing"], - agtech: ["agriculture"], foodtech: ["food"], - legaltech: ["legal"], hrtech: ["hr"], -}; -const CONTINENT = { - us: "na", ca: "na", mx: "na", - gb: "eu", de: "eu", fr: "eu", es: "eu", it: "eu", nl: "eu", se: "eu", ch: "eu", ie: "eu", pl: "eu", pt: "eu", be: "eu", at: "eu", dk: "eu", no: "eu", fi: "eu", - cn: "as", jp: "as", in: "as", sg: "as", kr: "as", hk: "as", il: "as", ae: "as", - br: "sa", ar: "sa", cl: "sa", co: "sa", - au: "oc", nz: "oc", - za: "af", ng: "af", ke: "af", eg: "af", -}; -// Rough country centroids (lat, lng) used as a fallback when an entity -// has an ISO2 country but no precise coordinates. Only the most common -// targets are populated; absent entries skip the Haversine path and -// fall back to ISO2 + continent. -const COUNTRY_CENTROID = { - us: [39.8, -98.6], ca: [56.1, -106.3], mx: [23.6, -102.5], - gb: [54.0, -2.0], de: [51.2, 10.4], fr: [46.6, 2.2], - es: [40.4, -3.7], it: [41.9, 12.5], nl: [52.1, 5.3], - se: [60.1, 18.6], ch: [46.8, 8.2], ie: [53.1, -7.7], - cn: [35.9, 104.2], jp: [36.2, 138.2], in: [20.6, 78.9], - sg: [1.35, 103.8], kr: [35.9, 127.8], il: [31.0, 34.9], - br: [-14.2, -51.9], au: [-25.3, 133.8], nz: [-40.9, 174.9], - za: [-30.6, 22.9], ng: [9.1, 8.7], -}; -// --------------------------------------------------------------------------- -// Cosine similarity (exported for the embedding-driven title_sim). -// --------------------------------------------------------------------------- -export function cosine(a, b) { - if (a.length !== b.length || !a.length) - return 0; - let dot = 0, na = 0, nb = 0; - for (let i = 0; i < a.length; i++) { - dot += a[i] * b[i]; - na += a[i] * a[i]; - nb += b[i] * b[i]; - } - if (!na || !nb) - return 0; - return Math.max(0, dot / (Math.sqrt(na) * Math.sqrt(nb))); -} -// --------------------------------------------------------------------------- -// Component scorers (each returns a value in [0, 1]). -// --------------------------------------------------------------------------- -export function scoreSeniority(entity, targets) { - if (!entity || !targets.length) { - return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: "no seniority data" }; - } - const e = entity.toLowerCase().trim(); - const ei = SENIORITY_INDEX[e]; - if (ei === undefined) { - return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: `unknown seniority "${entity}"` }; - } - let best = 0; - let bestTarget = ""; - for (const t of targets) { - const ti = SENIORITY_INDEX[t.toLowerCase().trim()]; - if (ti === undefined) - continue; - const d = Math.abs(ei - ti); - let v = 0; - if (d === 0) - v = 1.0; - else if (d === 1) - v = 0.6; - else if (d === 2) - v = 0.2; - if (v > best) { - best = v; - bestTarget = t; - } - } - return { - value: best, weight: DEFAULT_WEIGHTS.seniority, - reason: best === 1 ? `exact seniority match (${entity})` - : best > 0 ? `seniority ${entity} ≈ target ${bestTarget}` - : `seniority ${entity} too far from targets`, - }; -} -function stem(token) { - let t = token.toLowerCase().replace(/[^a-z]/g, ""); - if (t.endsWith("ing") && t.length > 5) - t = t.slice(0, -3); - else if (t.endsWith("ed") && t.length > 4) - t = t.slice(0, -2); - else if (t.endsWith("es") && t.length > 4) - t = t.slice(0, -2); - else if (t.endsWith("s") && t.length > 3) - t = t.slice(0, -1); - return t; -} -function tokenSet(s) { - if (!s) - return new Set(); - return new Set(s.split(/[\s,/&-]+/).map(stem).filter((t) => t.length > 1)); -} -export function scoreFunction(entityDept, targets) { - if (!entityDept || !targets.length) { - return { value: 0, weight: DEFAULT_WEIGHTS.function, reason: "no function/dept data" }; - } - const ent = tokenSet(entityDept); - let best = 0; - let bestTarget = ""; - for (const t of targets) { - const tgt = tokenSet(t); - if (!tgt.size) - continue; - let hits = 0; - for (const tok of tgt) - if (ent.has(tok)) - hits++; - const jacc = hits / Math.max(1, new Set([...ent, ...tgt]).size); - if (jacc > best) { - best = jacc; - bestTarget = t; - } - } - return { - value: Math.min(1, best * 1.5), - weight: DEFAULT_WEIGHTS.function, - reason: best > 0 ? `function "${entityDept}" overlaps "${bestTarget}"` : `function "${entityDept}" no overlap`, - }; -} -export function scoreIndustry(entityIndustries, targets) { - if (!entityIndustries.length || !targets.length) { - return { value: 0, weight: DEFAULT_WEIGHTS.industry, reason: "no industry data" }; - } - const tset = new Set(targets.map((t) => t.toLowerCase())); - let best = 0; - let bestNote = ""; - for (const ei of entityIndustries) { - const e = ei.toLowerCase(); - if (tset.has(e)) { - best = 1.0; - bestNote = `industry "${ei}" matches target`; - break; - } - const parents = INDUSTRY_PARENTS[e] ?? []; - for (const p of parents) { - if (tset.has(p)) { - if (best < 0.7) { - best = 0.7; - bestNote = `industry "${ei}" ⊂ target "${p}"`; - } - } - } - } - if (!bestNote) - bestNote = `industries [${entityIndustries.join(",")}] don't match targets`; - return { value: best, weight: DEFAULT_WEIGHTS.industry, reason: bestNote }; -} -export function scoreCompanySize(emp, minE, maxE) { - if (emp == null || (minE == null && maxE == null)) { - return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: "company size unknown" }; - } - const lo = minE ?? 0; - const hi = maxE ?? Number.POSITIVE_INFINITY; - if (emp >= lo && emp <= hi) { - return { value: 1.0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} in target [${lo}, ${maxE ?? "∞"}]` }; - } - const dLo = emp < lo ? (lo - emp) / Math.max(1, lo) : 0; - const dHi = emp > hi ? (emp - hi) / Math.max(1, hi) : 0; - const d = Math.max(dLo, dHi); - if (d <= 0.5) - return { value: 0.5, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} adjacent to target` }; - return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} outside target [${lo}, ${maxE ?? "∞"}]` }; -} -export function scoreStage(entityStages, targets) { - if (!entityStages.length || !targets.length) { - return { value: 0, weight: DEFAULT_WEIGHTS.stage, reason: "stage unknown" }; - } - let best = 0; - let note = ""; - for (const e of entityStages) { - const ei = STAGE_INDEX[e.toLowerCase().replace(/[-\s]/g, "_")]; - if (ei === undefined) - continue; - for (const t of targets) { - const ti = STAGE_INDEX[t.toLowerCase().replace(/[-\s]/g, "_")]; - if (ti === undefined) - continue; - const d = Math.abs(ei - ti); - let v = 0; - if (d === 0) - v = 1.0; - else if (d === 1) - v = 0.6; - if (v > best) { - best = v; - note = d === 0 ? `stage ${e} matches target` : `stage ${e} adjacent to ${t}`; - } - } - } - return { value: best, weight: DEFAULT_WEIGHTS.stage, reason: note || `stages [${entityStages.join(",")}] don't match targets` }; -} -// --------------------------------------------------------------------------- -// Geo: Haversine + exponential decay when coordinates are present; ISO2 -// + continent fallback otherwise. -// --------------------------------------------------------------------------- -const EARTH_KM = 6371; -export function haversineKm(lat1, lng1, lat2, lng2) { - const toRad = (x) => (x * Math.PI) / 180; - const dLat = toRad(lat2 - lat1); - const dLng = toRad(lng2 - lng1); - const a = Math.sin(dLat / 2) ** 2 + - Math.cos(toRad(lat1)) * Math.cos(toRad(lat2)) * Math.sin(dLng / 2) ** 2; - return 2 * EARTH_KM * Math.asin(Math.min(1, Math.sqrt(a))); -} -export function scoreGeo(input) { - const { entityIso2, targets } = input; - const hasCenter = input.centerLat != null && input.centerLng != null && (input.radiusKm ?? 0) > 0; - // Coordinate path: Haversine + exp(-d/radius) decay. - let entLat = input.entityLat ?? null; - let entLng = input.entityLng ?? null; - if ((entLat == null || entLng == null) && entityIso2) { - const c = COUNTRY_CENTROID[entityIso2.toLowerCase()]; - if (c) { - entLat = c[0]; - entLng = c[1]; - } - } - if (hasCenter && entLat != null && entLng != null) { - const d = haversineKm(input.centerLat, input.centerLng, entLat, entLng); - const v = Math.max(0, Math.min(1, Math.exp(-d / Math.max(1, input.radiusKm)))); - return { - value: v, - weight: DEFAULT_WEIGHTS.geo, - reason: `geo ${d.toFixed(0)}km from persona center (radius ${input.radiusKm}km) ⇒ ${v.toFixed(2)}`, - data: { distance_km: d, radius_km: input.radiusKm }, - }; - } - // ISO2 fallback. - if (!entityIso2 || !targets.length) { - return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: "geo unknown" }; - } - const e = entityIso2.toLowerCase(); - const t = targets.map((x) => x.toLowerCase()); - if (t.includes(e)) { - return { value: 1.0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} matches target` }; - } - const ec = CONTINENT[e]; - if (ec) { - for (const tc of t) - if (CONTINENT[tc] === ec) { - return { value: 0.5, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} shares region with target ${tc.toUpperCase()}` }; - } - } - return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} not in targets [${targets.join(",")}]` }; -} -// --------------------------------------------------------------------------- -// Aggregation + rationale. -// --------------------------------------------------------------------------- -export function aggregate(components) { - let sum = 0; - let wsum = 0; - for (const k of Object.keys(components)) { - const c = components[k]; - sum += c.value * c.weight; - wsum += c.weight; - } - return wsum > 0 ? sum / wsum : 0; -} -export function buildRationale(personaName, entityName, employerName, components, score) { - const pct = Math.round(score * 100); - const top = Object.entries(components) - .map(([k, c]) => ({ k, contribution: c.value * c.weight, reason: c.reason })) - .sort((a, b) => b.contribution - a.contribution) - .slice(0, 3) - .map((x) => x.reason) - .join("; "); - const who = entityName ?? "entity"; - const where = employerName ? ` at ${employerName}` : ""; - return `${who}${where} scores ${pct}% against persona "${personaName}". Top drivers: ${top}.`; -} -function arrFromJson(s) { - if (!s) - return []; - try { - const v = JSON.parse(s); - return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; - } - catch { - return []; - } -} -function objFromJson(s) { - if (!s) - return {}; - try { - const v = JSON.parse(s); - return v && typeof v === "object" && !Array.isArray(v) ? v : {}; - } - catch { - return {}; - } -} -export function extractTargets(row) { - const hard = objFromJson(row.hard_filters_json); - const stagesFromHard = Array.isArray(hard.stages) ? hard.stages.filter((x) => typeof x === "string") - : Array.isArray(hard.target_stage) ? hard.target_stage.filter((x) => typeof x === "string") - : []; - const center = hard.geo_center && typeof hard.geo_center === "object" ? hard.geo_center : {}; - const titles = arrFromJson(row.buyer_titles_json); - const seniority = arrFromJson(row.buyer_seniority_json); - const functions = arrFromJson(row.buyer_departments_json); - const industries = arrFromJson(row.industries_json); - const geos = arrFromJson(row.geos_json); - // Task #8 spec: title_sim must use ONLY structured persona target - // fields — no long-form notes (thesis, free text). Embedding here is - // restricted to titles + seniority + function so the component score - // stays explainable and reproducible. - const title_text = [ - titles.join(", "), - seniority.length ? `Seniority: ${seniority.join(", ")}` : "", - functions.length ? `Function: ${functions.join(", ")}` : "", - ].filter(Boolean).join(". "); - return { - title_text, - titles, - seniority, - functions, - industries, - size_min: row.size_min ?? null, - size_max: row.size_max ?? null, - stages: stagesFromHard, - geos, - geo_center_lat: typeof center.lat === "number" ? center.lat : null, - geo_center_lng: typeof center.lng === "number" ? center.lng : null, - geo_radius_km: typeof center.radius_km === "number" ? center.radius_km - : typeof hard.radius_km === "number" ? hard.radius_km : null, - }; -} diff --git a/apps/worker/test-dist-q/services/personas/kinds/_generic.js b/apps/worker/test-dist-q/services/personas/kinds/_generic.js deleted file mode 100644 index 7af6e0ef..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/_generic.js +++ /dev/null @@ -1,45 +0,0 @@ -// Task #3: Generic kind plugin used as the default for kinds that -// don't ship a bespoke matcher. Drives candidate selection from the -// taxonomy's `targets` (entity kind) + `roles` (entity_roles.role IN) -// and delegates scoring to the existing PersonaMatchingService scorer -// for person targets. Company/fund targets currently fall back to the -// legacy persona_matches/accounts/buyers code path via the dispatcher. -import { loadPersonEntity, scoreEntityForPersona as scorePersonForPersona } from "../../personaMatching"; -import { KINDS } from "./taxonomy"; -export function makeGenericPlugin(kind) { - const def = KINDS[kind]; - return { - kind, - defaultEntityFilter(_persona, opts) { - const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); - const offset = Math.max(0, opts?.offset ?? 0); - const binds = [def.targets]; - let sql = `SELECT DISTINCT e.id FROM u_entities e`; - if (def.roles.length) { - sql += ` JOIN entity_roles r ON r.entity_id = e.id`; - } - sql += ` WHERE e.kind = ? AND e.status = 'active'`; - if (def.roles.length) { - sql += ` AND r.role IN (${def.roles.map(() => "?").join(",")})`; - binds.push(...def.roles); - } - sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; - binds.push(limit, offset); - return { sql, binds }; - }, - async scoreEntity(env, persona, entityId) { - // Default behavior: only person targets are scored via the - // graph scorer. Fund/company targets are out of scope for the - // person-graph matcher and return null so the caller skips them. - if (def.targets !== "person") - return null; - const entity = await loadPersonEntity(env, entityId); - if (!entity) - return null; - return await scorePersonForPersona(env, persona, entity); - }, - explainMatch(entityId) { - return `kind=${kind} target=${def.targets}${def.roles.length ? " roles=" + def.roles.join("|") : ""} entity=${entityId}`; - }, - }; -} diff --git a/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js b/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js deleted file mode 100644 index a56a2ac2..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=academic_researcher. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const AcademicResearcherPlugin = makeGenericPlugin("academic_researcher"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/account_company.js b/apps/worker/test-dist-q/services/personas/kinds/account_company.js deleted file mode 100644 index 8072a299..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/account_company.js +++ /dev/null @@ -1,17 +0,0 @@ -// Task #3: account_company kind plugin (legacy "account" kind). -// -// Sales-side accounts live in the legacy `accounts` table and are -// scored via personas/score.ts + persona_matches (Task #46), not via -// the u_entities person graph. This plugin exists so the dispatcher -// can identify the kind, but defaultEntityFilter returns an empty -// candidate set — the legacy code path in routes/personas.ts owns -// account rescoring end-to-end. -export const accountCompanyPlugin = { - kind: "account_company", - defaultEntityFilter(_persona, _opts) { - // Legacy path: accounts are not in u_entities for persona matching. - return { sql: `SELECT id FROM u_entities WHERE 1 = 0`, binds: [] }; - }, - async scoreEntity(_env, _persona, _entityId) { return null; }, - explainMatch(entityId) { return `account_company (legacy accounts table): entity=${entityId}`; }, -}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/acquirer.js b/apps/worker/test-dist-q/services/personas/kinds/acquirer.js deleted file mode 100644 index 72f5b2c9..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/acquirer.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=acquirer. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const AcquirerPlugin = makeGenericPlugin("acquirer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js b/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js deleted file mode 100644 index d4d9efde..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=angel_individual. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const AngelIndividualPlugin = makeGenericPlugin("angel_individual"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js b/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js deleted file mode 100644 index 40cbde98..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=beta_tester. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const BetaTesterPlugin = makeGenericPlugin("beta_tester"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js b/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js deleted file mode 100644 index 6f8f008e..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js +++ /dev/null @@ -1,37 +0,0 @@ -// Task #3: buyer_person kind plugin (legacy "buyer" kind). -// -// Combines: (a) the new u_entities person-graph filter on role IN -// ('buyer','decision_maker','champion'), (b) delegation to the -// person-graph scorer. The legacy `buyers` table is still rescored -// by personas/rescore.ts under the hood; this plugin only governs -// the new u_entities-backed candidate list. -import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; -const ROLES = ["buyer", "decision_maker", "champion"]; -export const buyerPersonPlugin = { - kind: "buyer_person", - defaultEntityFilter(_persona, opts) { - const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); - const offset = Math.max(0, opts?.offset ?? 0); - const ph = ROLES.map(() => "?").join(","); - return { - // No role filter when entity_roles lacks buyer-flavored rows yet - // — fall back to all active person entities. Matches the legacy - // behavior so existing 'buyer' personas don't regress to empty. - sql: `SELECT e.id FROM u_entities e - WHERE e.kind = 'person' AND e.status = 'active' - AND ( - EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id AND r.role IN (${ph})) - OR NOT EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id) - ) - ORDER BY e.id LIMIT ? OFFSET ?`, - binds: [...ROLES, limit, offset], - }; - }, - async scoreEntity(env, persona, entityId) { - const entity = await loadPersonEntity(env, entityId); - if (!entity) - return null; - return await scoreEntityForPersona(env, persona, entity); - }, - explainMatch(entityId) { return `buyer_person match: entity=${entityId}`; }, -}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js b/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js deleted file mode 100644 index 3eb5cc89..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=channel_partner. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const ChannelPartnerPlugin = makeGenericPlugin("channel_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js b/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js deleted file mode 100644 index c3786ce0..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=co_founder_match. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const CoFounderMatchPlugin = makeGenericPlugin("co_founder_match"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/competitor.js b/apps/worker/test-dist-q/services/personas/kinds/competitor.js deleted file mode 100644 index 3801e793..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/competitor.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=competitor. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const CompetitorPlugin = makeGenericPlugin("competitor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/design_partner.js b/apps/worker/test-dist-q/services/personas/kinds/design_partner.js deleted file mode 100644 index 04200099..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/design_partner.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=design_partner. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const DesignPartnerPlugin = makeGenericPlugin("design_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js b/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js deleted file mode 100644 index ecebae4a..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=engineering_hire. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const EngineeringHirePlugin = makeGenericPlugin("engineering_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js b/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js deleted file mode 100644 index a45b4674..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=executive_hire. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const ExecutiveHirePlugin = makeGenericPlugin("executive_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/founder.js b/apps/worker/test-dist-q/services/personas/kinds/founder.js deleted file mode 100644 index 6ef92773..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/founder.js +++ /dev/null @@ -1,68 +0,0 @@ -// Task #3: founder kind plugin. -// -// Surfaces hint fields `founded_count`, `prior_exits`, `domain_expertise`. -// Match: entities with role IN ('founder','ceo','co_founder'). Hint -// numerics are validated by the form; we re-validate here so a stale -// or hand-edited persona can't crash the scorer. -import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; -const ROLES = ["founder", "ceo", "co_founder"]; -function readHint(persona, field) { - if (!persona.hard_filters_json) - return null; - try { - const j = JSON.parse(persona.hard_filters_json); - const v = j?.hints?.[field]; - return v == null ? null : String(v); - } - catch { - return null; - } -} -function splitCsv(v) { - return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; -} -export const founderPlugin = { - kind: "founder", - defaultEntityFilter(persona, opts) { - const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); - const offset = Math.max(0, opts?.offset ?? 0); - // Hints become hard predicates so they actually narrow the - // candidate set (not just decorate the UI). Roles in entity_roles - // act as a tag space — domain expertise becomes a 'domain:' - // tag, prior_exits / founded_count map to 'exits:N+' / 'founded:N+' - // synthetic tags that the enrichment pipeline emits per founder. - const domains = splitCsv(readHint(persona, "domain_expertise")); - const foundedMin = parseInt(readHint(persona, "founded_count") ?? "", 10); - const exitsMin = parseInt(readHint(persona, "prior_exits") ?? "", 10); - const binds = []; - const rolePh = ROLES.map(() => "?").join(","); - let sql = `SELECT DISTINCT e.id FROM u_entities e - JOIN entity_roles r ON r.entity_id = e.id - WHERE e.kind = 'person' AND e.status = 'active' - AND r.role IN (${rolePh})`; - binds.push(...ROLES); - if (domains.length) { - const ph = domains.map(() => "?").join(","); - sql += ` AND EXISTS (SELECT 1 FROM entity_roles rd WHERE rd.entity_id = e.id AND rd.role IN (${ph}))`; - binds.push(...domains.map((d) => `domain:${d}`)); - } - if (Number.isFinite(foundedMin) && foundedMin > 0) { - sql += ` AND EXISTS (SELECT 1 FROM entity_roles rf WHERE rf.entity_id = e.id AND rf.role = ?)`; - binds.push(`founded:${foundedMin}+`); - } - if (Number.isFinite(exitsMin) && exitsMin > 0) { - sql += ` AND EXISTS (SELECT 1 FROM entity_roles re WHERE re.entity_id = e.id AND re.role = ?)`; - binds.push(`exits:${exitsMin}+`); - } - sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; - binds.push(limit, offset); - return { sql, binds }; - }, - async scoreEntity(env, persona, entityId) { - const entity = await loadPersonEntity(env, entityId); - if (!entity) - return null; - return await scoreEntityForPersona(env, persona, entity); - }, - explainMatch(entityId) { return `founder match: entity=${entityId}`; }, -}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js b/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js deleted file mode 100644 index 6bcd934d..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=fractional_executive. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const FractionalExecutivePlugin = makeGenericPlugin("fractional_executive"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js b/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js deleted file mode 100644 index 0c548ec6..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=government_grant_officer. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const GovernmentGrantOfficerPlugin = makeGenericPlugin("government_grant_officer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/index.js b/apps/worker/test-dist-q/services/personas/kinds/index.js deleted file mode 100644 index ed940044..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/index.js +++ /dev/null @@ -1,71 +0,0 @@ -// Task #3: persona-kind plugin registry + dispatcher entrypoint. -// -// The PersonaMatchingService reads the persona's kind, calls -// getPluginFor(kind), and delegates to defaultEntityFilter / scoreEntity. -// Kinds without a bespoke plugin file fall back to the generic plugin -// driven by the taxonomy's `roles` array (see _generic.ts). -import { ALL_KIND_KEYS, KINDS, resolveKind } from "./taxonomy"; -import { investorPersonPlugin } from "./investor_person"; -import { investorFirmPlugin } from "./investor_firm"; -import { venturePartnerPlugin } from "./venture_partner"; -import { founderPlugin } from "./founder"; -import { accountCompanyPlugin } from "./account_company"; -import { buyerPersonPlugin } from "./buyer_person"; -import { AngelIndividualPlugin } from "./angel_individual"; -import { LimitedPartnerPlugin } from "./limited_partner"; -import { CoFounderMatchPlugin } from "./co_founder_match"; -import { ExecutiveHirePlugin } from "./executive_hire"; -import { EngineeringHirePlugin } from "./engineering_hire"; -import { FractionalExecutivePlugin } from "./fractional_executive"; -import { ChannelPartnerPlugin } from "./channel_partner"; -import { IntegrationPartnerPlugin } from "./integration_partner"; -import { DesignPartnerPlugin } from "./design_partner"; -import { BetaTesterPlugin } from "./beta_tester"; -import { JournalistAnalystPlugin } from "./journalist_analyst"; -import { ThoughtLeaderPlugin } from "./thought_leader"; -import { AcademicResearcherPlugin } from "./academic_researcher"; -import { GovernmentGrantOfficerPlugin } from "./government_grant_officer"; -import { RegulatorPlugin } from "./regulator"; -import { PolicyAdvisorPlugin } from "./policy_advisor"; -import { ServiceProviderPlugin } from "./service_provider"; -import { AcquirerPlugin } from "./acquirer"; -import { CompetitorPlugin } from "./competitor"; -const REGISTRY_MAP = { - account_company: accountCompanyPlugin, - buyer_person: buyerPersonPlugin, - investor_person: investorPersonPlugin, - investor_firm: investorFirmPlugin, - venture_partner: venturePartnerPlugin, - founder: founderPlugin, - angel_individual: AngelIndividualPlugin, - limited_partner: LimitedPartnerPlugin, - co_founder_match: CoFounderMatchPlugin, - executive_hire: ExecutiveHirePlugin, - engineering_hire: EngineeringHirePlugin, - fractional_executive: FractionalExecutivePlugin, - channel_partner: ChannelPartnerPlugin, - integration_partner: IntegrationPartnerPlugin, - design_partner: DesignPartnerPlugin, - beta_tester: BetaTesterPlugin, - journalist_analyst: JournalistAnalystPlugin, - thought_leader: ThoughtLeaderPlugin, - academic_researcher: AcademicResearcherPlugin, - government_grant_officer: GovernmentGrantOfficerPlugin, - regulator: RegulatorPlugin, - policy_advisor: PolicyAdvisorPlugin, - service_provider: ServiceProviderPlugin, - acquirer: AcquirerPlugin, - competitor: CompetitorPlugin, -}; -// Sanity check: every declared kind has a plugin file. If a future -// kind is added to taxonomy without a wrapper, surface it at boot. -for (const k of ALL_KIND_KEYS) { - if (!REGISTRY_MAP[k]) - throw new Error(`missing plugin file for kind=${k}`); -} -export function getPluginFor(rawKind) { - const k = resolveKind(rawKind) ?? "account_company"; - return REGISTRY_MAP[k]; -} -export { KINDS, ALL_KIND_KEYS, resolveKind }; -export * from "./taxonomy"; diff --git a/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js b/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js deleted file mode 100644 index 194ec5ec..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=integration_partner. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const IntegrationPartnerPlugin = makeGenericPlugin("integration_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js b/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js deleted file mode 100644 index c4af4a4f..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js +++ /dev/null @@ -1,64 +0,0 @@ -// Task #3: investor_firm kind plugin. -// -// Targets entities where kind='fund' (or 'firm' legacy) AND -// entity_roles.role='investor_firm'. The person-graph scorer doesn't -// apply to funds directly; matching here is structural (role + kind -// + AUM/stage hints) and we expose a deterministic surface so the -// dispatcher can present candidates even without per-entity scoring. -// Read a hint value from hard_filters_json.hints.. -function readHint(persona, field) { - if (!persona.hard_filters_json) - return null; - try { - const j = JSON.parse(persona.hard_filters_json); - const v = j?.hints?.[field]; - return typeof v === "string" && v ? v : null; - } - catch { - return null; - } -} -// Split a comma-separated hint value into trimmed tokens. -function splitCsv(v) { - return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; -} -export const investorFirmPlugin = { - kind: "investor_firm", - defaultEntityFilter(persona, opts) { - const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); - const offset = Math.max(0, opts?.offset ?? 0); - // Roles in entity_roles act as a tag space. When the persona - // specifies aum_band or stage_focus hints, we require that the - // fund carries the corresponding tag-role (e.g. 'aum:$1B-$5B' or - // 'stage:seed'). This way the hints actually narrow the candidate - // set rather than just decorating the UI. - const aum = readHint(persona, "aum_band"); // e.g. "$1B-$5B" - const stages = splitCsv(readHint(persona, "stage_focus")); // e.g. ["seed","series_a"] - const binds = []; - let sql = `SELECT DISTINCT e.id FROM u_entities e - JOIN entity_roles r ON r.entity_id = e.id - WHERE e.status = 'active' - AND e.kind IN ('fund','firm') - AND r.role = 'investor_firm'`; - if (aum) { - sql += ` AND EXISTS (SELECT 1 FROM entity_roles ra WHERE ra.entity_id = e.id AND ra.role = ?)`; - binds.push(`aum:${aum}`); - } - if (stages.length) { - const ph = stages.map(() => "?").join(","); - sql += ` AND EXISTS (SELECT 1 FROM entity_roles rs WHERE rs.entity_id = e.id AND rs.role IN (${ph}))`; - binds.push(...stages.map((s) => `stage:${s}`)); - } - sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; - binds.push(limit, offset); - return { sql, binds }; - }, - async scoreEntity(_env, _persona, _entityId) { - // Structural-only match for funds — no per-entity scoring at this - // tier. Dispatcher treats null as "candidate present, score 0.5". - return null; - }, - explainMatch(entityId) { - return `investor_firm structural match: entity=${entityId} role=investor_firm kind=fund|firm`; - }, -}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_person.js b/apps/worker/test-dist-q/services/personas/kinds/investor_person.js deleted file mode 100644 index dfec68e4..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/investor_person.js +++ /dev/null @@ -1,14 +0,0 @@ -// Task #3: investor_person kind plugin. -// -// Acceptance criteria: matches entities where type='person' AND -// entity_roles.role IN ('investor','vc','gp','partner_at_firm'), -// then filters by firm size / stage / sector criteria via the -// existing person-graph scorer (which already considers employer -// sectors / stages / employees from career_history + entity_summary). -import { makeGenericPlugin } from "./_generic"; -// The generic plugin's defaultEntityFilter already picks up the -// taxonomy's role list ['investor','vc','gp','partner_at_firm'] and -// the person scorer already weighs employer sector / stage / size. -// We export it under a stable name so the registry can swap in a -// bespoke implementation later without touching the registry wiring. -export const investorPersonPlugin = makeGenericPlugin("investor_person"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js b/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js deleted file mode 100644 index 086a2293..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=journalist_analyst. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const JournalistAnalystPlugin = makeGenericPlugin("journalist_analyst"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js b/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js deleted file mode 100644 index ace8b1a2..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=limited_partner. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const LimitedPartnerPlugin = makeGenericPlugin("limited_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js b/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js deleted file mode 100644 index 39706ce5..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=policy_advisor. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const PolicyAdvisorPlugin = makeGenericPlugin("policy_advisor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/regulator.js b/apps/worker/test-dist-q/services/personas/kinds/regulator.js deleted file mode 100644 index 0219b79f..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/regulator.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=regulator. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const RegulatorPlugin = makeGenericPlugin("regulator"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/service_provider.js b/apps/worker/test-dist-q/services/personas/kinds/service_provider.js deleted file mode 100644 index 4abb6751..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/service_provider.js +++ /dev/null @@ -1,7 +0,0 @@ -// Task #3: thin plugin wrapper for kind=service_provider. Delegates to the -// generic plugin which drives candidate selection from taxonomy -// roles+targets and scoring via the person-graph scorer for person -// targets. Kept as a discrete file so the plugin-per-kind contract -// is satisfied and future per-kind customization has a home. -import { makeGenericPlugin } from "./_generic"; -export const ServiceProviderPlugin = makeGenericPlugin("service_provider"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js b/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js deleted file mode 100644 index 981f0019..00000000 --- a/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js +++ /dev/null @@ -1,86 +0,0 @@ -// Task #3: Expand persona kinds taxonomy. -// -// Single source of truth for the persona-kind taxonomy. Both the form -// (consumed via GET /api/personas/taxonomy) and the matcher dispatcher -// read from this module. No taxonomy duplication in HTML or in plugin -// code — plugins reference KINDS[kind] to discover their group, label, -// allowed criteria sections, and required hint fields. -// Hint-field metadata for the form (label, type, options). -export const HINTS = { - subtype: { label: "Subtype", type: "select", options: ["lawyer", "banker", "operator", "politician", "scout", "advisor", "board_member"] }, - aum_band: { label: "AUM band", type: "select", options: ["<$50M", "$50M-$250M", "$250M-$1B", "$1B-$5B", ">$5B"] }, - stage_focus: { label: "Stage focus", type: "text", placeholder: "pre_seed, seed, series_a" }, - founded_count: { label: "Companies founded (min)", type: "number" }, - prior_exits: { label: "Prior exits (min)", type: "number" }, - domain_expertise: { label: "Domain expertise", type: "text", placeholder: "fintech, dev_tools" }, -}; -const COMMON = ["geography", "industry", "signals", "tuning"]; -const PERSON_BASE = [...COMMON, "buyer_profile"]; -const COMPANY_BASE = ["sizing", ...COMMON, "tech_stack"]; -export const KINDS_LIST = [ - // ---- Sales - { kind: "account_company", group: "Sales", label: "Account (company)", sections: COMPANY_BASE, hints: [], targets: "company", roles: [] }, - { kind: "buyer_person", group: "Sales", label: "Buyer (person)", sections: PERSON_BASE, hints: [], targets: "person", roles: ["buyer", "decision_maker", "champion"] }, - // ---- Capital - { kind: "investor_firm", group: "Capital", label: "Investor firm", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band", "stage_focus"], targets: "fund", roles: ["investor_firm"] }, - { kind: "investor_person", group: "Capital", label: "Investor (person)", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["stage_focus"], targets: "person", roles: ["investor", "vc", "gp", "partner_at_firm"] }, - { kind: "angel_individual", group: "Capital", label: "Angel investor", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["angel", "investor"] }, - { kind: "limited_partner", group: "Capital", label: "Limited partner", sections: ["geography", "signals", "tuning"], hints: ["aum_band"], targets: "person", roles: ["limited_partner", "lp"] }, - { kind: "venture_partner", group: "Capital", label: "Venture partner", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["subtype", "domain_expertise"], targets: "person", roles: ["venture_partner", "advisor", "scout"] }, - // ---- People - { kind: "founder", group: "People", label: "Founder", sections: ["geography", "industry", "signals", "tuning"], hints: ["founded_count", "prior_exits", "domain_expertise"], targets: "person", roles: ["founder", "ceo"] }, - { kind: "co_founder_match", group: "People", label: "Co-founder match", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["founder", "engineer", "designer"] }, - { kind: "executive_hire", group: "People", label: "Executive hire", sections: PERSON_BASE, hints: ["domain_expertise"], targets: "person", roles: ["executive", "vp", "c_suite"] }, - { kind: "engineering_hire", group: "People", label: "Engineering hire", sections: ["geography", "tech_stack", "signals", "buyer_profile", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["engineer", "ic"] }, - { kind: "fractional_executive", group: "People", label: "Fractional executive", sections: PERSON_BASE, hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["fractional", "advisor", "executive"] }, - // ---- Partnerships - { kind: "channel_partner", group: "Partnerships", label: "Channel partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["partner", "reseller"] }, - { kind: "integration_partner", group: "Partnerships", label: "Integration partner", sections: ["sizing", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["partner", "integration"] }, - { kind: "design_partner", group: "Partnerships", label: "Design partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["customer", "prospect"] }, - { kind: "beta_tester", group: "Partnerships", label: "Beta tester", sections: ["geography", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["customer", "prospect", "beta"] }, - // ---- Influence (no tech_stack — content/coverage focused) - { kind: "journalist_analyst", group: "Influence", label: "Journalist / analyst", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["journalist", "analyst", "press"] }, - { kind: "thought_leader", group: "Influence", label: "Thought leader", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["influencer", "thought_leader", "speaker"] }, - { kind: "academic_researcher", group: "Influence", label: "Academic researcher", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["researcher", "academic", "professor"] }, - // ---- Public Sector - { kind: "government_grant_officer", group: "Public Sector", label: "Government grant officer", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["government", "grant_officer", "program_officer"] }, - { kind: "regulator", group: "Public Sector", label: "Regulator", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["regulator", "agency_official"] }, - { kind: "policy_advisor", group: "Public Sector", label: "Policy advisor", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["policy_advisor", "staffer", "aide"] }, - // ---- Operational - { kind: "service_provider", group: "Operational", label: "Service provider", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "company", roles: ["vendor", "service_provider", "agency"] }, - { kind: "acquirer", group: "Operational", label: "Acquirer", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band"], targets: "company", roles: ["acquirer", "strategic"] }, - { kind: "competitor", group: "Operational", label: "Competitor", sections: ["sizing", "geography", "industry", "tech_stack", "tuning"], hints: [], targets: "company", roles: ["competitor"] }, -]; -export const KINDS = Object.fromEntries(KINDS_LIST.map((k) => [k.kind, k])); -export const ALL_KIND_KEYS = KINDS_LIST.map((k) => k.kind); -// Legacy values from before Task #3 — map to the closest new kind so -// existing personas keep working without a backfill migration. -export const LEGACY_KIND_MAP = { - account: "account_company", - buyer: "buyer_person", -}; -export function resolveKind(raw) { - if (!raw) - return null; - if (KINDS[raw]) - return raw; - if (LEGACY_KIND_MAP[raw]) - return LEGACY_KIND_MAP[raw]; - return null; -} -export function isValidKind(raw) { - return resolveKind(raw) !== null; -} -// Grouped view for the form's grouped