From f7e74c90011be966a6b9a4f4fc660641e394a65a Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 22:48:03 +0000 Subject: [PATCH 1/7] Fix four dashboard paths that loaded fine and showed nothing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four faults with one signature: the page rendered, nothing errored, and the content was empty — so each read as "we have no data on this entity" rather than "this link is wrong". 1. /dashboard/profile/ ignored the param most pages link it with. profile-tab.js read only `?entity=`. Seven links across six pages — ops-garbage-review.js, predictions.html (twice), watchlists.html, power-nodes.html, dossiers.html, ops-quality.html — pass `?id=`, and every one of them landed on "No entity selected." Fixed in the reader rather than at the seven call sites, so links written the same way in future also work. 2. The Leads list sent a legacy id to an entity-keyed page. Every name linked to /dashboard/people/?id=, but that page feeds the value to /api/profilers/:entity_id/*, which is keyed on u_entities. A legacy integer id never matches a uuid, so every name on the page opened a profile with every panel blank. GET /api/leads now returns the mapped `entity_id` — the listing already joins entity_legacy_map for the role badges, so it costs one more correlated subquery — and the name links there when it exists, or to the lead detail page (which is keyed on exactly the id we have) when it does not. 3. Bulk "select page" was dead on two of the four pages that offer it. bulk-bar.js resolved #ads-bulk-header-check once in init(). On leads and accounts that cell is static HTML, so it worked. On investors and companies it is emitted inside the async row render — absent at init(), and replaced wholesale on every non-append load. The checkbox still ticked, so it looked like it had worked. Now bound by delegation like the bar's own buttons. The double-click reset timer is cleared rather than stacked; the old code queued a new 2.5 s timer per click, so an earlier one could fire mid-gesture and cancel the second click. 4. The deep health sweep was publicly callable. The Access allow-list covers /api/health because it is a cheap liveness probe, but `/deep` shares that router: 18 binding probes, three COUNT(*) scans of error_log, and a live fetchPage() with an 8 s budget that walks every fetcher tier — including the metered Browser Rendering tier when the cheaper tiers escalate. Unauthenticated, anyone could drive that in a loop. accessGuard is now registered for both /deep mounts, before the health router so it runs first. Verified against the repo's own Hono: /health and /api/health still answer 200, both deep paths answer 401. The documented operator workflow is a browser session, which carries the cookie the guard reads, so it is unaffected; the ops checklist now says so and points uptime monitors at the cheap probe. The site has no JS test runner, so test/site_deeplinks.test.mjs asserts the first three contracts statically over the files — the approach ci_wrangler_version.test.mjs already takes for the workflow YAML — and access_guard.test.mjs gains the fourth. Each was verified to fail with its fix reverted. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/site/assets/js/bulk-bar.js | 32 ++++--- apps/site/assets/js/leads.js | 19 +++- apps/site/assets/js/profile-tab.js | 17 +++- apps/worker/package.json | 2 +- apps/worker/src/index.ts | 11 +++ apps/worker/src/routes/leads.ts | 14 ++- apps/worker/test/access_guard.test.mjs | 24 +++++ apps/worker/test/site_deeplinks.test.mjs | 107 +++++++++++++++++++++++ docs/cloudflare-operations-checklist.md | 10 +++ 9 files changed, 220 insertions(+), 16 deletions(-) create mode 100644 apps/worker/test/site_deeplinks.test.mjs diff --git a/apps/site/assets/js/bulk-bar.js b/apps/site/assets/js/bulk-bar.js index 78533a32..429cf5ab 100644 --- a/apps/site/assets/js/bulk-bar.js +++ b/apps/site/assets/js/bulk-bar.js @@ -130,7 +130,6 @@ var selection = new Map(); // id -> true var allMatchingMode = false; // true after 2nd header click var currentSignature = cfg.getFilterSignature ? cfg.getFilterSignature() : ""; - var headerCheck = document.getElementById("ads-bulk-header-check"); // Rehydrate selection on init iff the persisted signature matches // the current filter signature. Drop otherwise. @@ -211,15 +210,28 @@ } } - if (headerCheck) { - var headerClicks = 0; - headerCheck.addEventListener("click", function () { - headerClicks += 1; - if (headerClicks === 1) { selectPage(headerCheck.checked); } - else { headerClicks = 0; selectAllMatching(); } - setTimeout(function () { headerClicks = 0; }, 2500); - }); - } + // Delegated, not bound to the element. On the investors and companies + // pages the header checkbox is emitted inside the async row render, so it + // does not exist when init() runs and it is replaced wholesale on every + // non-append load. A direct listener resolved at init() caught neither + // case: "select page" and "select all matching" were dead on both pages, + // and the checkbox still ticked, so it looked like it had worked. + // (It did work on leads and accounts, whose header cell is static HTML — + // which is why the two pages behaved differently for the same code.) + var headerClicks = 0; + var headerResetTimer = null; + document.addEventListener("click", function (e) { + var hc = e.target && e.target.closest ? e.target.closest("#ads-bulk-header-check") : null; + if (!hc) return; + headerClicks += 1; + if (headerClicks === 1) { selectPage(hc.checked); } + else { headerClicks = 0; selectAllMatching(); } + // Cleared rather than stacked: the old code queued a fresh 2.5 s reset + // on every click, so an earlier timer could fire between the two clicks + // of a deliberate double-click and reset the counter mid-gesture. + if (headerResetTimer) clearTimeout(headerResetTimer); + headerResetTimer = setTimeout(function () { headerClicks = 0; }, 2500); + }); // Rebind row checks after each list refresh — pages call this manually // (or use a MutationObserver as a fallback). diff --git a/apps/site/assets/js/leads.js b/apps/site/assets/js/leads.js index 18633d06..ab3c419c 100644 --- a/apps/site/assets/js/leads.js +++ b/apps/site/assets/js/leads.js @@ -72,6 +72,23 @@ }).join(""); } + // Where a lead's name should go. + // + // This used to be an unconditional /dashboard/people/?id=. That page + // hands the value to /api/profilers/:entity_id/*, which is keyed on + // u_entities — and `l.id` is the legacy `leads` primary key, a different id + // space entirely. So the link never resolved: every name on the Leads page + // opened a person profile with every panel empty, which reads as "we have + // no data on this person" rather than "this link is wrong". + // + // The listing now carries `entity_id` when the lead has been mapped into + // u_entities. Rows that have not been mapped yet go to the lead detail + // page, which is keyed on exactly the id we have. + function nameHref(l) { + if (l.entity_id) return "/dashboard/people/?id=" + encodeURIComponent(l.entity_id); + return "/dashboard/lead/?id=" + encodeURIComponent(l.id); + } + function render(items) { var tbody = document.getElementById("ads-leads-tbody"); if (!items.length) { @@ -81,7 +98,7 @@ tbody.innerHTML = items.map(function (l) { return '' + '' - + '' + esc(l.name || "(no name)") + '' + rolesBadges(l.roles) + '' + + '' + esc(l.name || "(no name)") + '' + rolesBadges(l.roles) + '' + '' + esc(l.org || "—") + '' + '' + esc(l.email || "—") + '' + '' + esc(l.status || "—") + '' diff --git a/apps/site/assets/js/profile-tab.js b/apps/site/assets/js/profile-tab.js index 368def98..65acffb7 100644 --- a/apps/site/assets/js/profile-tab.js +++ b/apps/site/assets/js/profile-tab.js @@ -4,7 +4,8 @@ // window.ADS.Profile.mount({ rootId, entityId }) — embed mode // window.ADS.Profile.mountStandalone() — /dashboard/profile/ // -// Resolves entity via ?entity= or ?table=&ref= (same convention as DD/News). +// Resolves entity via ?entity= (or ?id=) or ?table=&ref= (same +// convention as DD/News). (function () { if (window.ADS && window.ADS.Profile) return; @@ -27,9 +28,19 @@ sec_rel: "Secular ↔ Religious", }; + // `?id=` is accepted alongside `?entity=` because seven navigation entry + // points across six pages link here with `?id=` — ops-garbage-review.js, + // predictions.html (twice), watchlists.html, power-nodes.html, + // dossiers.html and ops-quality.html. Reading only `entity` made every one + // of those land on "No entity selected." Fixing the reader rather than the + // seven call sites also covers any future link written the same way. function qs() { var p = new URLSearchParams(window.location.search); - return { entity: p.get("entity"), table: p.get("table"), ref: p.get("ref") }; + return { + entity: p.get("entity") || p.get("id"), + table: p.get("table"), + ref: p.get("ref"), + }; } function esc(s) { return String(s == null ? "" : s).replace(/[&<>"']/g, function (c) { @@ -399,7 +410,7 @@ if (!entityId) { if (titleEl) titleEl.textContent = "Profile"; var host = document.getElementById("ads-profile-root"); - if (host) host.innerHTML = '
No entity selected. Pass ?entity= or ?table=&ref=.
'; + if (host) host.innerHTML = '
No entity selected. Pass ?entity=, ?id= or ?table=&ref=.
'; return; } if (titleEl) titleEl.textContent = "Loading…"; diff --git a/apps/worker/package.json b/apps/worker/package.json index 30567bc8..2fbdc575 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index 87447e4e..167f2b80 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -127,6 +127,17 @@ api.use( allowHeaders: ["Content-Type", "Cf-Access-Jwt-Assertion", "Idempotency-Key"], }), ); +// The public allow-list covers the CHEAP liveness probe, and only that. +// `/deep` shares the same router, so mounting the router publicly also +// published a readiness sweep that runs 18 binding probes, three COUNT(*) +// scans of error_log, and a live outbound fetchPage() with an 8 s budget +// that walks every fetcher tier — including the metered Browser Rendering +// tier when the cheaper tiers escalate. Unauthenticated, that is a cost +// amplifier anyone can drive in a loop, and nothing about "health is +// public" was ever meant to grant it. Guard the deep sweep specifically; +// registering the middleware BEFORE the router is what makes it run first. +api.use("/health/deep", accessGuard); +api.use("/api/health/deep", accessGuard); api.route("/health", health); api.route("/api/health", health); api.route("/api/webhooks/campaigns", campaignsWebhook); diff --git a/apps/worker/src/routes/leads.ts b/apps/worker/src/routes/leads.ts index f3561307..32efebe4 100644 --- a/apps/worker/src/routes/leads.ts +++ b/apps/worker/src/routes/leads.ts @@ -81,7 +81,19 @@ leads.get("/", async (c) => { `SELECT ${RICH_COLUMNS}, (SELECT json_group_array(r.role) FROM entity_legacy_map m JOIN entity_roles r ON r.entity_id = m.entity_id - WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id) AS roles_json + WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id) AS roles_json, + -- The unified id for this lead, when one exists. The Leads list + -- linked each name to /dashboard/people/?id=, but that + -- page feeds the value straight to /api/profilers/:entity_id/*, + -- which is keyed on u_entities. A legacy integer id never matches + -- a uuid, so every name on the page opened an empty profile. The + -- list already joins entity_legacy_map for the role badges, so + -- carrying the id costs one more correlated subquery and lets the + -- page link to a person page that resolves — or fall back to the + -- lead detail page for rows that have no entity yet. + (SELECT m.entity_id FROM entity_legacy_map m + WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id + LIMIT 1) AS entity_id FROM leads ${whereSql} ${orderSql} LIMIT ?`, ).bind(...binds, limit); const r = await stmt.all(); diff --git a/apps/worker/test/access_guard.test.mjs b/apps/worker/test/access_guard.test.mjs index ba6eb746..060a4cf4 100644 --- a/apps/worker/test/access_guard.test.mjs +++ b/apps/worker/test/access_guard.test.mjs @@ -79,6 +79,30 @@ test("/health and /api/health are both public per spec", () => { assert.ok(apiHealthIdx < guardIdx, "/api/health must be mounted before accessGuard (public allow-list per task #2)"); }); +// The allow-list covers the CHEAP liveness probe. `/deep` lives on the same +// router, so mounting that router publicly also published a readiness sweep +// that runs 18 binding probes, three COUNT(*) scans of error_log, and a live +// outbound fetchPage() that walks every fetcher tier — the metered Browser +// Rendering tier included. Unauthenticated that is a cost amplifier anyone +// can drive in a loop. Verified against the repo's own Hono: the cheap +// probes still answer 200 and both deep paths answer 401. +test("the deep health sweep is NOT public, on either mount", () => { + const guardIdx = src.search(/api\.use\(\s*"\/api\/\*"\s*,\s*accessGuard\s*\)/); + for (const path of ["/health/deep", "/api/health/deep"]) { + const re = new RegExp(`api\\.use\\(\\s*"${path.replace(/\//g, "\\/")}"\\s*,\\s*accessGuard\\s*\\)`); + const idx = src.search(re); + assert.ok(idx > -1, `${path} has no accessGuard — the deep sweep is publicly callable`); + // Hono runs handlers in registration order and stops at the first that + // returns, so the guard only fires if it is registered before the router. + const routeIdx = src.search( + new RegExp(`api\\.route\\(\\s*"${path.replace(/\/deep$/, "").replace(/\//g, "\\/")}"`), + ); + assert.ok(routeIdx > -1, `mount for ${path} not found`); + assert.ok(idx < routeIdx, `${path} guard must be registered before the health router`); + assert.ok(idx < guardIdx || guardIdx === -1, `${path} guard should sit with the public mounts`); + } +}); + test("onError returns a sanitized envelope in production (no Error.stack, no raw internal message)", () => { // Source-level assertion: the production branch of onError must // build a `{ error: { code, message } }` envelope, never include diff --git a/apps/worker/test/site_deeplinks.test.mjs b/apps/worker/test/site_deeplinks.test.mjs new file mode 100644 index 00000000..ca83e238 --- /dev/null +++ b/apps/worker/test/site_deeplinks.test.mjs @@ -0,0 +1,107 @@ +// Dashboard deep links that pointed at pages which could not read them. +// +// Three separate faults, all with the same signature: the page loaded, the +// chrome rendered, and the content was empty. Nothing errored, so each one +// read as "we have no data on this entity" rather than "this link is wrong". +// +// 1. Seven links across six pages open /dashboard/profile/?id=. +// profile-tab.js read only `?entity=`, so all seven showed +// "No entity selected." +// 2. leads.js linked every name to /dashboard/people/?id=. +// That page feeds the value to /api/profilers/:entity_id/*, which is +// keyed on u_entities — a legacy integer id can never match a uuid. +// 3. bulk-bar.js resolved #ads-bulk-header-check once at init(). On +// investors and companies that element is emitted inside the async row +// render, so it did not exist yet; "select page" and "select all +// matching" were dead there while working on leads and accounts. +// +// The site has no JS test runner, so these are static assertions over the +// files — the same approach ci_wrangler_version.test.mjs takes for the +// workflow YAML. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, readdirSync, statSync } from "node:fs"; +import { join, dirname, relative } from "node:path"; +import { fileURLToPath } from "node:url"; + +const SITE = join(dirname(fileURLToPath(import.meta.url)), "../../site"); + +function walk(dir, out = []) { + for (const name of readdirSync(dir)) { + if (name === "node_modules" || name === "vendor" || name === "_site") continue; + const p = join(dir, name); + if (statSync(p).isDirectory()) walk(p, out); + else if (/\.(js|html)$/.test(name)) out.push(p); + } + return out; +} + +const FILES = walk(SITE); +const read = (rel) => readFileSync(join(SITE, rel), "utf8"); + +test("the site tree is where this test thinks it is", () => { + assert.ok(FILES.length > 50, `expected the dashboard sources, found ${FILES.length} files`); + assert.ok(FILES.some((f) => f.endsWith("assets/js/profile-tab.js"))); +}); + +// ---- 1. /dashboard/profile/ deep links --------------------------------- + +test("profile-tab.js reads every query param the site links to it with", () => { + const src = read("assets/js/profile-tab.js"); + // The params the reader honours, from the qs() destructure. + const qs = src.match(/function qs\(\)\s*\{[\s\S]*?\n \}/); + assert.ok(qs, "qs() not found — the shape of profile-tab.js changed"); + + const used = new Set(); + for (const f of FILES) { + for (const m of readFileSync(f, "utf8").matchAll(/\/dashboard\/profile\/\?([a-z_]+)=/g)) { + used.add(m[1]); + } + } + assert.ok(used.size > 0, "no /dashboard/profile/ links found — the scan is broken"); + + const unread = [...used].filter((p) => !new RegExp(`p\\.get\\("${p}"\\)`).test(qs[0])); + assert.deepEqual(unread, [], + `pages deep-link /dashboard/profile/ with these params, and qs() ignores them, ` + + `so the page renders "No entity selected": ${unread.join(", ")}`); +}); + +// ---- 2. the Leads list id space ---------------------------------------- + +test("leads.js does not send a legacy leads id to an entity-keyed page", () => { + const src = read("assets/js/leads.js"); + assert.ok(!/\/dashboard\/people\/\?id=['"]\s*\+\s*(?:esc|encodeURIComponent)\(l\.id\)/.test(src), + "leads.js links a leads.id at /dashboard/people/, which resolves it against u_entities"); + assert.ok(/l\.entity_id/.test(src), + "leads.js should link the unified entity id when the listing carries one"); +}); + +test("the leads listing actually returns the entity_id leads.js links with", () => { + const route = readFileSync(join(SITE, "../worker/src/routes/leads.ts"), "utf8"); + assert.ok(/AS entity_id/.test(route), + "GET /api/leads must select entity_id or leads.js has nothing to link to"); +}); + +// ---- 3. the bulk-bar header checkbox ----------------------------------- + +test("bulk-bar binds the header checkbox by delegation, not by init-time lookup", () => { + const src = read("assets/js/bulk-bar.js"); + assert.ok(!/getElementById\(\s*["']ads-bulk-header-check["']\s*\)/.test(src), + "resolving the header checkbox at init() misses the pages that render it asynchronously"); + assert.ok(/closest\(\s*["']#ads-bulk-header-check["']\s*\)/.test(src), + "bulk-bar should match the header checkbox on a delegated click"); +}); + +test("the pages that render the header checkbox asynchronously still declare it", () => { + // If a page stops emitting the id the delegated handler is inert — silently, + // which is the failure mode this whole file exists to catch. + const emitters = FILES.filter((f) => /id="ads-bulk-header-check"/.test(readFileSync(f, "utf8"))) + .map((f) => relative(SITE, f)).sort(); + for (const expected of [ + "assets/js/companies.js", "assets/js/investors.js", + "dashboard/accounts.html", "dashboard/leads/index.html", + ]) { + assert.ok(emitters.includes(expected), `${expected} no longer renders #ads-bulk-header-check`); + } +}); diff --git a/docs/cloudflare-operations-checklist.md b/docs/cloudflare-operations-checklist.md index e3e864c4..649d9d2d 100644 --- a/docs/cloudflare-operations-checklist.md +++ b/docs/cloudflare-operations-checklist.md @@ -174,6 +174,16 @@ Then, in an operator browser session (Access blocks curl): `GET https://api.aidatasignal.com/api/health/deep` probes every binding, and `/ops/system-health/` shows the latest health snapshot and open incidents. +`/api/health/deep` requires that Access session by design. It runs 18 binding +probes, three `COUNT(*)` scans of `error_log` and a live `fetchPage()` that can +escalate to the metered Browser Rendering tier, so it is not something to point +an uptime monitor at. Use the cheap public probe for that: + +``` +GET https://api.aidatasignal.com/health # public, 200 + a D1 ping +GET https://api.aidatasignal.com/api/health # same handler, same cost +``` + ## Known state (2026-09-04) The Cloudflare account connected to the Claude Code session that produced From 7604056c1c0017e927287bcbb197a1a29215ea87 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 22:51:47 +0000 Subject: [PATCH 2/7] Persona matching could not score above 0.90, on any entity, ever MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit scoreCompanySize carries weight 0.10 and returns 0 with the reason "company size unknown" when it has no headcount. loadEmployerFacts asked `facts` for `org.headcount`, `org.employees`, `company.employees` and `company.headcount` — four spellings, and nothing in the worker has ever written one of them. So the component was zero for every entity against every persona: a hard ceiling of 0.90 on the score, and a permanent "company size unknown" line in the rationale the dashboard shows to explain why someone matched. The predicate that exists is bare `employees`. It is what the registry declares in entities/profile-predicates.ts and what secEdgar/persist.ts already writes, so this converges on the name in use rather than adding a fifth spelling. The four unused ones are kept in the IN list — harmless, and they resolve if a future writer picks one. Reading it is not enough on its own, so the loop is closed end to end: - `accounts.employees` was the one sizing column the account dual-write dropped, so no account entity carried a headcount fact at all. It now writes `employees`. - backfillAccounts names its columns explicitly and did not name that one, so the bulk path — the one that actually populates the graph — would have kept dropping it while the single-record insert path kept it. Both now carry it. - personaMatchTrigger's RELEVANT_PREDICATES set gains `employees`, or a fresh headcount fact would never re-score the entity it belongs to. Checked and deliberately not changed: loadEntityCoords reads six lat/lng predicates that nothing writes either, but scoreGeo falls back to a country centroid and then to an ISO2 comparison, so that one degrades as designed rather than silently zeroing a component. test/persona_headcount.test.mjs asserts written, read and re-triggered together, because fixing two of the three would look right and still score 0.90. Verified it fails with either half reverted. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/package.json | 2 +- apps/worker/src/entities/backfill.ts | 2 +- apps/worker/src/entities/dualwrite.ts | 8 ++ .../src/services/personaMatchTrigger.ts | 3 + apps/worker/src/services/personaMatching.ts | 13 ++- apps/worker/test/persona_headcount.test.mjs | 81 +++++++++++++++++++ 6 files changed, 106 insertions(+), 3 deletions(-) create mode 100644 apps/worker/test/persona_headcount.test.mjs diff --git a/apps/worker/package.json b/apps/worker/package.json index 2fbdc575..37c4ab02 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/entities/backfill.ts b/apps/worker/src/entities/backfill.ts index 8b160d3b..c4b38751 100644 --- a/apps/worker/src/entities/backfill.ts +++ b/apps/worker/src/entities/backfill.ts @@ -79,7 +79,7 @@ export async function backfillAccounts(env: Env, offset = 0, limit = BATCH): Pro `SELECT id, name, legal_name, website, domain, industry, industries_json, hq_country_iso2, hq_region, hq_city, funding_stage, linkedin_url, twitter_handle, github_org, crunchbase_url, - fit_score, intent_score + fit_score, intent_score, employees FROM accounts ORDER BY created_at LIMIT ? OFFSET ?`, ).bind(limit + 1, offset).all>(); const rows = r.results ?? []; diff --git a/apps/worker/src/entities/dualwrite.ts b/apps/worker/src/entities/dualwrite.ts index 83045e4f..32858121 100644 --- a/apps/worker/src/entities/dualwrite.ts +++ b/apps/worker/src/entities/dualwrite.ts @@ -100,6 +100,7 @@ interface AccountLikeInput { crunchbase_url?: string | null; fit_score?: number | null; intent_score?: number | null; + employees?: number | null; } interface BuyerLikeInput { @@ -408,6 +409,13 @@ export async function syncAccountToEntity(env: Env, a: AccountLikeInput, source { predicate: "funding_stage", value_text: a.funding_stage ?? null }, { predicate: "fit_max_score", value_number: numOrNull(a.fit_score) }, { predicate: "intent_score", value_number: numOrNull(a.intent_score) }, + // `accounts.employees` was the one sizing column dualwrite dropped, so + // no account entity carried a headcount fact and persona matching + // scored every one of them "company size unknown". `employees` is the + // predicate the registry declares (entities/profile-predicates.ts) and + // the one secEdgar/persist.ts already writes, so this converges on the + // name that exists rather than adding a fourth spelling. + { predicate: "employees", value_number: numOrNull(a.employees) }, ]; await insertFactsBatch(env, entityId, patches, source, "scrape"); await Promise.all([ diff --git a/apps/worker/src/services/personaMatchTrigger.ts b/apps/worker/src/services/personaMatchTrigger.ts index 7b862f35..1b66a711 100644 --- a/apps/worker/src/services/personaMatchTrigger.ts +++ b/apps/worker/src/services/personaMatchTrigger.ts @@ -16,6 +16,9 @@ const RELEVANT_PREDICATES = new Set([ "person.location.country", "person.location.city", "location.country", "location.city", "employer", "person.employer", + // `employees` is the spelling that actually gets written; without it a + // fresh headcount fact never re-scored the entity it belongs to. + "employees", "org.headcount", "org.employees", "company.employees", "company.headcount", "org.sector", "sector", diff --git a/apps/worker/src/services/personaMatching.ts b/apps/worker/src/services/personaMatching.ts index a9408c05..5b5242cb 100644 --- a/apps/worker/src/services/personaMatching.ts +++ b/apps/worker/src/services/personaMatching.ts @@ -65,8 +65,19 @@ async function loadEmployerFacts(env: Env, employerId: string): Promise<{ const sum = await env.DB.prepare( `SELECT display_name, country_iso2, sectors_csv, stages_csv FROM entity_summary WHERE entity_id = ?`, ).bind(employerId).first<{ display_name: string | null; country_iso2: string | null; sectors_csv: string | null; stages_csv: string | null }>(); + // `employees` is first because it is the only one of these that anything + // writes: it is the predicate the registry declares + // (entities/profile-predicates.ts) and the one secEdgar/persist.ts and the + // account dual-write emit. The four `org.*` / `company.*` spellings below + // were the entire list, and no writer has ever produced one — so + // `employees` came back null for every entity and scoreCompanySize + // returned its "company size unknown" zero every time. With a weight of + // 0.10 that put a hard ceiling of 0.90 on every persona match, and made + // "company size unknown" a permanent line in the rationale the dashboard + // shows to explain why someone matched. The unused spellings are kept so a + // future writer picking one still resolves. const hc = await env.DB.prepare( - `SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`, + `SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('employees','org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`, ).bind(employerId).first<{ value_number: number | null }>(); if (!sum && !hc) return null; return { diff --git a/apps/worker/test/persona_headcount.test.mjs b/apps/worker/test/persona_headcount.test.mjs new file mode 100644 index 00000000..e70e52b7 --- /dev/null +++ b/apps/worker/test/persona_headcount.test.mjs @@ -0,0 +1,81 @@ +// Persona matching could never score above 0.90, and always explained why +// with a line that was not true. +// +// scoreCompanySize carries weight 0.10 and returns 0 with the reason +// "company size unknown" when it has no headcount. personaMatching.ts asked +// facts for `org.headcount`, `org.employees`, `company.employees` and +// `company.headcount` — four predicate spellings that no writer in the +// worker has ever produced. So the component was zero for every entity +// against every persona: a hard ceiling of 0.90, and a permanent +// "company size unknown" in the rationale the dashboard shows to explain +// the match. +// +// The predicate that exists is bare `employees`: it is what the registry +// declares (entities/profile-predicates.ts) and what secEdgar/persist.ts +// writes. The fix converges on that name rather than inventing a fifth. +// +// This asserts the loop is closed end to end — written, read, and +// re-triggered on change — because closing two of the three would have +// looked fixed and still scored 0.90. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { join, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = join(dirname(fileURLToPath(import.meta.url)), ".."); +const src = (rel) => readFileSync(join(ROOT, rel), "utf8"); + +const PREDICATE = "employees"; + +test("the registry declares the predicate this test is about", () => { + const reg = src("src/entities/profile-predicates.ts"); + assert.match(reg, new RegExp(`predicate:\\s*"${PREDICATE}"`), + "profile-predicates.ts no longer declares `employees` — pick the new canonical name and update all three sites"); +}); + +test("something writes a headcount fact", () => { + const writers = ["src/services/secEdgar/persist.ts", "src/entities/dualwrite.ts"]; + const writing = writers.filter((f) => new RegExp(`predicate:\\s*"${PREDICATE}"`).test(src(f))); + assert.deepEqual(writing, writers, + `these should each write a \`${PREDICATE}\` fact; without a writer the read below is decorative`); +}); + +test("the account dual-write and its backfill both carry the column", () => { + // The insert path reads the full row, the backfill path names its columns. + // Omitting it there left the bulk path — the one that actually populates + // the graph — dropping headcount while the single-record path kept it. + assert.match(src("src/entities/dualwrite.ts"), /employees\?:\s*number\s*\|\s*null/, + "AccountLikeInput must accept employees or the dual-write silently binds null"); + const backfill = src("src/entities/backfill.ts"); + const select = backfill.match(/SELECT[\s\S]*?FROM accounts/); + assert.ok(select, "backfillAccounts SELECT not found"); + assert.match(select[0], /\bemployees\b/, + "backfillAccounts must select employees or the bulk path drops it"); +}); + +test("persona matching reads the predicate that is written", () => { + const matching = src("src/services/personaMatching.ts"); + const inList = matching.match(/predicate IN \(([^)]*)\)[^`]*value_number IS NOT NULL/); + assert.ok(inList, "the headcount lookup in loadEmployerFacts changed shape"); + assert.ok(inList[1].includes(`'${PREDICATE}'`), + `loadEmployerFacts asks for ${inList[1]} — none of which any writer produces, ` + + `so scoreCompanySize returns 0 for every entity and caps every match at 0.90`); +}); + +test("a new headcount fact re-triggers the match", () => { + const trigger = src("src/services/personaMatchTrigger.ts"); + const set = trigger.match(/RELEVANT_PREDICATES = new Set\(\[([\s\S]*?)\]\)/); + assert.ok(set, "RELEVANT_PREDICATES not found"); + assert.ok(set[1].includes(`"${PREDICATE}"`), + "without this a fresh headcount fact never re-scores the entity it belongs to"); +}); + +test("company_size still carries the weight that made this matter", () => { + const scorers = src("src/services/personaMatchingScorers.ts"); + const m = scorers.match(/company_size:\s*([0-9.]+)/); + assert.ok(m, "company_size weight not found"); + assert.ok(Number(m[1]) > 0, + "if company_size were zero-weighted the bug would be cosmetic; it is not"); +}); From 1e0b2fa832898bdb322c752b515343c10f6abf57 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 22:55:56 +0000 Subject: [PATCH 3/7] Make an idle buyer-signal crawler distinguishable from a working one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The seven ATS sources — greenhouse, lever, ashby, workable, recruitee, personio, smartrecruiters — each select accounts whose meta_json carries an operator-set key (greenhouse_board, lever_company, ashby_company, workable_account, recruitee_company, personio_company, smartrecruiters_company). Nothing in the worker sets any of them automatically. That is by design — the keys are operator-seeded — but it means that on a deployment where nobody has seeded one, every source returns zero rows and records `0 emitted, ok`: byte-for-byte the same run row as a source that fetched a hundred boards and found nothing new. Each source now returns `seeded_accounts` and `boards_fetched`, counted from the loop rather than hardcoded, so "nothing to scan" and "scanned, nothing new" are different rows. That alone would have changed nothing, because the channel it travels on was never connected. CrawlResult.meta is documented as "per-source counters appended to crawler_runs.meta_json" and the column has existed since migration 161, but runCrawl.ts never read the field and its finalize UPDATE never wrote the column — so any counters a source returned were dropped on the floor. runCrawl now reads and persists them, stringifying defensively so a source that returns something unserialisable cannot be the reason a run records no outcome at all. The crawlers console gains a Notes column that renders those counters; meta_json already reached the client via SELECT *, nothing displayed it. This does not build ATS detection — no source key is now set that was not set before. It makes the absence visible instead of silent, which is the part that was wrong. test/crawler_run_meta.test.mjs asserts all four links: the sources report, the counters are derived, runCrawl persists them, and the console renders them. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/site/dashboard/crawlers.html | 20 ++++- apps/worker/package.json | 2 +- apps/worker/src/prospects/runCrawl.ts | 22 ++++- apps/worker/src/prospects/sources/ashby.ts | 14 +++- .../src/prospects/sources/greenhouse.ts | 14 +++- apps/worker/src/prospects/sources/lever.ts | 14 +++- apps/worker/src/prospects/sources/personio.ts | 14 +++- .../worker/src/prospects/sources/recruitee.ts | 14 +++- .../src/prospects/sources/smartrecruiters.ts | 14 +++- apps/worker/src/prospects/sources/workable.ts | 14 +++- apps/worker/test/crawler_run_meta.test.mjs | 82 +++++++++++++++++++ 11 files changed, 210 insertions(+), 14 deletions(-) create mode 100644 apps/worker/test/crawler_run_meta.test.mjs diff --git a/apps/site/dashboard/crawlers.html b/apps/site/dashboard/crawlers.html index b9769305..35a18e90 100644 --- a/apps/site/dashboard/crawlers.html +++ b/apps/site/dashboard/crawlers.html @@ -41,10 +41,11 @@

Recent runs

Inserted Skipped Accts ↑ + Notes Error - Loading… + Loading… @@ -54,6 +55,20 @@

Recent runs

(async function () { const tbody = document.querySelector("#ads-crawlers-table tbody"); const runsBody = document.querySelector("#ads-crawler-runs tbody"); + // crawler_runs.meta_json holds the per-source counters. Rendering them is + // the difference between "this source had nothing to scan" and "this source + // scanned everything and found nothing new" — both of which otherwise show + // as a run with 0 emitted and status ok. The seven ATS sources are seeded + // from accounts.meta_json keys that nothing sets automatically, so + // seeded_accounts=0 is the answer to "why is this always empty?". + function runNotes(metaJson) { + if (!metaJson) return ""; + var m; + try { m = JSON.parse(metaJson); } catch (e) { return ""; } + if (!m || typeof m !== "object") return ""; + return Object.keys(m).map(function (k) { return k + "=" + m[k]; }).join(" · "); + } + function fmt(s) { return s ? new Date(s).toLocaleString() : "—"; } function esc(s) { return String(s == null ? "" : s).replace(/[&<>"']/g, function (c) { return ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" })[c]; }); } async function refresh() { @@ -87,11 +102,12 @@

Recent runs

${esc(x.signals_inserted)} ${esc(x.signals_skipped)} +${esc(x.accounts_created)} / ${esc(x.accounts_resolved)} + ${esc(runNotes(x.meta_json))} ${esc(x.error || "")} `).join(""); } catch (e) { - runsBody.innerHTML = `Failed to load: ${esc(e && e.message ? e.message : e)}`; + runsBody.innerHTML = `Failed to load: ${esc(e && e.message ? e.message : e)}`; } } document.addEventListener("change", async (e) => { diff --git a/apps/worker/package.json b/apps/worker/package.json index 37c4ab02..f07c4347 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/prospects/runCrawl.ts b/apps/worker/src/prospects/runCrawl.ts index 24b785fc..f9a2ef3e 100644 --- a/apps/worker/src/prospects/runCrawl.ts +++ b/apps/worker/src/prospects/runCrawl.ts @@ -51,10 +51,17 @@ export async function runSource(env: Env, mod: SourceModule, opts?: { force?: bo let drafts: SignalEventDraft[] = []; let nextCursor: string | null | undefined = cursor; let crawlError: string | undefined; + // CrawlResult.meta is documented as "per-source counters appended to + // crawler_runs.meta_json" and the column has existed since migration 161, + // but nothing ever read the field — every source's counters were dropped on + // the floor. That is what left a source with no work to do and a source that + // scanned everything and found nothing new both recorded as `0 events, ok`. + let crawlMeta: Record | undefined; try { const r = await mod.crawl({ env, cursor, maxEvents: MAX_EVENTS_PER_RUN, accountId: opts?.accountId }); drafts = (r.events ?? []).slice(0, MAX_EVENTS_PER_RUN); nextCursor = r.cursor === undefined ? cursor : r.cursor; + crawlMeta = r.meta; } catch (e) { crawlError = (e as Error).message; } @@ -131,14 +138,21 @@ export async function runSource(env: Env, mod: SourceModule, opts?: { force?: bo if (nextCursor) await setCursor(env, mod.slug, nextCursor); const status: RunOutcome["status"] = crawlError ? "error" : (skipped > 0 && inserted === 0 ? "partial" : "ok"); - await finalize(env, runId, status, { events: drafts.length, inserted, skipped, created, resolved, cursor: nextCursor ?? null, error: crawlError }); + await finalize(env, runId, status, { events: drafts.length, inserted, skipped, created, resolved, cursor: nextCursor ?? null, error: crawlError, meta: crawlMeta }); return { runId, source: mod.slug, status, events_emitted: drafts.length, signals_inserted: inserted, signals_skipped: skipped, accounts_created: created, accounts_resolved: resolved, error: crawlError }; } -async function finalize(env: Env, runId: string, status: string, m: { events: number; inserted: number; skipped: number; created: number; resolved: number; cursor: string | null; error?: string }): Promise { +async function finalize(env: Env, runId: string, status: string, m: { events: number; inserted: number; skipped: number; created: number; resolved: number; cursor: string | null; error?: string; meta?: Record }): Promise { + // Serialised defensively: a source is free to put anything in `meta`, and a + // value that cannot be stringified (a cycle, a BigInt) must not be the + // reason a run never records its outcome at all. + let metaJson: string | null = null; + if (m.meta && Object.keys(m.meta).length) { + try { metaJson = JSON.stringify(m.meta); } catch { metaJson = null; } + } await env.DB.prepare( - `UPDATE crawler_runs SET finished_at = ?, status = ?, events_emitted = ?, signals_inserted = ?, signals_skipped = ?, accounts_created = ?, accounts_resolved = ?, cursor = ?, error = ? WHERE id = ?`, - ).bind(new Date().toISOString(), status, m.events, m.inserted, m.skipped, m.created, m.resolved, m.cursor, m.error ?? null, runId) + `UPDATE crawler_runs SET finished_at = ?, status = ?, events_emitted = ?, signals_inserted = ?, signals_skipped = ?, accounts_created = ?, accounts_resolved = ?, cursor = ?, error = ?, meta_json = ? WHERE id = ?`, + ).bind(new Date().toISOString(), status, m.events, m.inserted, m.skipped, m.created, m.resolved, m.cursor, m.error ?? null, metaJson, runId) .run().catch((e) => console.warn("crawler_runs finalize failed", e.message)); } diff --git a/apps/worker/src/prospects/sources/ashby.ts b/apps/worker/src/prospects/sources/ashby.ts index 1f2569e5..e929359c 100644 --- a/apps/worker/src/prospects/sources/ashby.ts +++ b/apps/worker/src/prospects/sources/ashby.ts @@ -23,13 +23,16 @@ const mod: SourceModule = { const rows = await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%ashby_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).ashby_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://jobs.ashbyhq.com/${encodeURIComponent(company)}.json`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: AshbyResp = {}; try { parsed = JSON.parse(res.body) as AshbyResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "ashby", res.body, "json"); @@ -50,7 +53,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `ashby_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/greenhouse.ts b/apps/worker/src/prospects/sources/greenhouse.ts index 0f836a10..a9e50d41 100644 --- a/apps/worker/src/prospects/sources/greenhouse.ts +++ b/apps/worker/src/prospects/sources/greenhouse.ts @@ -35,12 +35,15 @@ const mod: SourceModule = { LIMIT 100`, ).all(); let newest = since; + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let token = ""; try { token = String((JSON.parse(r.meta_json ?? "{}") as Record).greenhouse_board ?? ""); } catch { /* skip */ } if (!token) continue; + seeded += 1; const fetched = await fetchBoard(ctx.env, token); if (!fetched) continue; + boardsFetched += 1; const r2_key = await archiveRaw(ctx.env, "greenhouse", fetched.raw, "json"); const fresh = fetched.jobs.filter((j) => Date.parse(j.updated_at) > since); // Cluster: >= 5 new postings inside this run = hiring_burst. @@ -60,7 +63,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `greenhouse_board` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/lever.ts b/apps/worker/src/prospects/sources/lever.ts index 311f7ab9..0c0e8296 100644 --- a/apps/worker/src/prospects/sources/lever.ts +++ b/apps/worker/src/prospects/sources/lever.ts @@ -21,13 +21,16 @@ const mod: SourceModule = { const rows = await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%lever_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).lever_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://api.lever.co/v0/postings/${encodeURIComponent(company)}?mode=json`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let postings: Posting[] = []; try { postings = JSON.parse(res.body) as Posting[]; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "lever", res.body, "json"); @@ -47,7 +50,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? String(newest) : ctx.cursor }; + return { + events, + cursor: newest > since ? String(newest) : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `lever_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/personio.ts b/apps/worker/src/prospects/sources/personio.ts index 8d8e5dc2..776ec266 100644 --- a/apps/worker/src/prospects/sources/personio.ts +++ b/apps/worker/src/prospects/sources/personio.ts @@ -53,13 +53,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%personio_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).personio_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://${encodeURIComponent(company)}.jobs.personio.de/xml`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/xml" }); if (!res || !res.ok) continue; + boardsFetched += 1; const r2_key = await archiveRaw(ctx.env, "personio", res.body, "xml"); const positions = parsePositions(res.body); const fresh = positions.filter((p) => { @@ -82,7 +85,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `personio_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/recruitee.ts b/apps/worker/src/prospects/sources/recruitee.ts index 60fd329f..df40dab8 100644 --- a/apps/worker/src/prospects/sources/recruitee.ts +++ b/apps/worker/src/prospects/sources/recruitee.ts @@ -38,13 +38,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%recruitee_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).recruitee_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://${encodeURIComponent(company)}.recruitee.com/api/offers/`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: RtResp = {}; try { parsed = JSON.parse(res.body) as RtResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "recruitee", res.body, "json"); @@ -70,7 +73,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `recruitee_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/smartrecruiters.ts b/apps/worker/src/prospects/sources/smartrecruiters.ts index f775cb28..c5dfa103 100644 --- a/apps/worker/src/prospects/sources/smartrecruiters.ts +++ b/apps/worker/src/prospects/sources/smartrecruiters.ts @@ -38,13 +38,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%smartrecruiters_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let slug = ""; try { slug = String((JSON.parse(r.meta_json ?? "{}") as Record).smartrecruiters_company ?? ""); } catch { /* skip */ } if (!slug) continue; + seeded += 1; const url = `https://api.smartrecruiters.com/v1/companies/${encodeURIComponent(slug)}/postings?limit=100`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: SrResp = {}; try { parsed = JSON.parse(res.body) as SrResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "smartrecruiters", res.body, "json"); @@ -69,7 +72,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `smartrecruiters_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/workable.ts b/apps/worker/src/prospects/sources/workable.ts index 76172894..50545873 100644 --- a/apps/worker/src/prospects/sources/workable.ts +++ b/apps/worker/src/prospects/sources/workable.ts @@ -37,13 +37,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%workable_account%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let slug = ""; try { slug = String((JSON.parse(r.meta_json ?? "{}") as Record).workable_account ?? ""); } catch { /* skip */ } if (!slug) continue; + seeded += 1; const url = `https://apply.workable.com/api/v3/accounts/${encodeURIComponent(slug)}/jobs`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: WkResp = {}; try { parsed = JSON.parse(res.body) as WkResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "workable", res.body, "json"); @@ -68,7 +71,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `workable_account` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/test/crawler_run_meta.test.mjs b/apps/worker/test/crawler_run_meta.test.mjs new file mode 100644 index 00000000..aa385cf5 --- /dev/null +++ b/apps/worker/test/crawler_run_meta.test.mjs @@ -0,0 +1,82 @@ +// A buyer-signal source with nothing to scan looked exactly like one that +// scanned everything and found nothing new. +// +// The seven ATS sources (greenhouse, lever, ashby, workable, recruitee, +// personio, smartrecruiters) each read an operator-set key off +// accounts.meta_json — greenhouse_board, lever_company, ashby_company, +// workable_account, recruitee_company, personio_company, +// smartrecruiters_company. Nothing in the worker sets any of them +// automatically, so on a fresh deployment every one of those SELECTs returns +// zero rows and the run records `0 emitted, ok`. Identical to a healthy run. +// +// Two things were wrong, and fixing either alone leaves the operator blind: +// +// * The sources reported no counters at all. +// * CrawlResult.meta is documented as "per-source counters appended to +// crawler_runs.meta_json", and the column has existed since migration +// 161 — but runCrawl.ts never read the field and its UPDATE never wrote +// the column, so any counters a source did return were discarded. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, readdirSync } from "node:fs"; +import { join, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = join(dirname(fileURLToPath(import.meta.url)), ".."); +const src = (rel) => readFileSync(join(ROOT, rel), "utf8"); + +const ATS = [ + "greenhouse", "lever", "ashby", "workable", + "recruitee", "personio", "smartrecruiters", +]; + +test("all seven ATS sources are still present", () => { + const files = readdirSync(join(ROOT, "src/prospects/sources")); + for (const s of ATS) assert.ok(files.includes(`${s}.ts`), `${s}.ts missing`); +}); + +test("every ATS source reports how much it had to scan", () => { + const missing = ATS.filter((s) => { + const f = src(`src/prospects/sources/${s}.ts`); + return !/meta:\s*\{[^}]*seeded_accounts/.test(f); + }); + assert.deepEqual(missing, [], + `these report no counters, so "nothing seeded" and "nothing new" are the ` + + `same run row: ${missing.join(", ")}`); +}); + +test("the counters are derived, not hardcoded", () => { + // A literal `seeded_accounts: 0` would satisfy the check above and tell an + // operator nothing. + for (const s of ATS) { + const f = src(`src/prospects/sources/${s}.ts`); + assert.match(f, /let seeded = 0, boardsFetched = 0;/, `${s}: counters not declared`); + assert.match(f, /\bseeded \+= 1;/, `${s}: seeded is never incremented`); + assert.match(f, /\bboardsFetched \+= 1;/, `${s}: boardsFetched is never incremented`); + } +}); + +test("runCrawl persists the meta it is handed", () => { + const f = src("src/prospects/runCrawl.ts"); + assert.match(f, /crawlMeta = r\.meta;/, + "runCrawl must read CrawlResult.meta — the field was documented and ignored"); + assert.match(f, /UPDATE crawler_runs SET[^`]*meta_json = \?/, + "the finalize UPDATE must write meta_json or the counters are dropped on the floor"); +}); + +test("the crawler_runs column the counters land in exists", () => { + const dir = join(ROOT, "migrations"); + const sql = readdirSync(dir).filter((f) => f.endsWith(".sql")) + .map((f) => readFileSync(join(dir, f), "utf8")).join("\n"); + const block = sql.match(/CREATE TABLE IF NOT EXISTS crawler_runs\s*\(([\s\S]*?)\n\);/); + assert.ok(block, "crawler_runs not found in the migrations"); + assert.match(block[1], /\bmeta_json\b/, "crawler_runs has no meta_json column"); +}); + +test("the crawlers console renders the counters", () => { + const page = readFileSync(join(ROOT, "../site/dashboard/crawlers.html"), "utf8"); + assert.match(page, /function runNotes\(/, + "meta_json reaches the client via SELECT * but nothing rendered it"); + assert.match(page, /runNotes\(x\.meta_json\)/, "runNotes is defined but not called"); +}); From 45efaed804e802fa39c7fb89aa8d8b65628bd0b0 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 22:58:49 +0000 Subject: [PATCH 4/7] Founder diligence could not link a founder through facts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit getFoundersOf() unions two lookups: a `founder.company_founded` fact pointing at the company, and career_history rows whose role_title contains "founder". The fact branch matched on `value_entity_id`, but the only writer of that predicate — crawler/profileWorkflows/founder.ts — stores the company as free text in `value_text`, because at extraction time it has a company name off a bio page and no resolved entity id. So the fact branch returned nothing, every time. It did not fail loudly: the union silently reduced to the career_history path, and where that path had no row with a resolved organization_entity_id the founder checks reported "no founders on record" rather than "we could not link them". The lookup now also matches the company's display_name against a trimmed, lower-cased value_text, with the value_entity_id branch kept first because that is the correct shape and will match once an extractor resolves the company. Blank names are excluded explicitly — without that, TRIM('') = TRIM(' ') joins every blank-named fact to every blank-named company. test/diligence_founder_link.test.mjs lifts the SQL out of the source and runs it against the real migrations, so an equality that cannot match is caught by executing it rather than by reading it — which is how this got here. Verified the name case fails against the old query. Checked and deliberately not changed: `company.privacy_policy_url` and `company.customer_logo` also have no writer, but those checks return cautionResult("No privacy policy URL on record.") and needsHuman("no_customer_logos_recorded"). Those are visible, correct outcomes for absent data — honest degradation working as designed, not the silent-zero this commit fixes. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/package.json | 2 +- .../src/services/diligence/checks/founders.ts | 25 ++++- .../test/diligence_founder_link.test.mjs | 106 ++++++++++++++++++ 3 files changed, 129 insertions(+), 4 deletions(-) create mode 100644 apps/worker/test/diligence_founder_link.test.mjs diff --git a/apps/worker/package.json b/apps/worker/package.json index f07c4347..8f1b272a 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/services/diligence/checks/founders.ts b/apps/worker/src/services/diligence/checks/founders.ts index ffa1a69d..b633a694 100644 --- a/apps/worker/src/services/diligence/checks/founders.ts +++ b/apps/worker/src/services/diligence/checks/founders.ts @@ -10,10 +10,29 @@ async function getFoundersOf(env: import("../../../types").Env, companyEntityId: // founders mirrored as facts (founder.company_founded) or via career role 'founder' const out = new Set(); try { + // The name fallback is not belt-and-braces: it is the only branch that + // currently matches. The single writer of this predicate — the founder + // profile workflow (crawler/profileWorkflows/founder.ts) — stores the + // company as free text in value_text, because at extraction time it has a + // company NAME off a bio page and no resolved entity. Matching only on + // value_entity_id therefore returned nothing, every time, and the whole + // founder-diligence section quietly fell back to the career_history path. + // The value_entity_id branch is kept first because it is the correct + // shape and will match once an extractor resolves the company. const r = await env.DB.prepare( - `SELECT entity_id FROM facts - WHERE predicate = 'founder.company_founded' AND value_entity_id = ? AND is_current = 1`, - ).bind(companyEntityId).all<{ entity_id: string }>(); + `SELECT f.entity_id FROM facts f + WHERE f.predicate = 'founder.company_founded' + AND f.is_current = 1 + AND ( + f.value_entity_id = ? + OR (f.value_entity_id IS NULL + AND f.value_text IS NOT NULL + AND TRIM(f.value_text) <> '' + AND LOWER(TRIM(f.value_text)) = ( + SELECT LOWER(TRIM(u.display_name)) FROM u_entities u WHERE u.id = ? + )) + )`, + ).bind(companyEntityId, companyEntityId).all<{ entity_id: string }>(); for (const row of r.results ?? []) out.add(row.entity_id); } catch { /* table may differ */ } try { diff --git a/apps/worker/test/diligence_founder_link.test.mjs b/apps/worker/test/diligence_founder_link.test.mjs new file mode 100644 index 00000000..d4072a92 --- /dev/null +++ b/apps/worker/test/diligence_founder_link.test.mjs @@ -0,0 +1,106 @@ +// The founder-diligence section could not find founders through facts. +// +// getFoundersOf() unions two lookups: a `founder.company_founded` fact +// pointing at the company, and career_history rows whose role_title contains +// "founder". The fact branch matched on `value_entity_id`. The only writer of +// that predicate — crawler/profileWorkflows/founder.ts — stores the company +// as free TEXT in value_text, because at extraction time it has a company +// name off a bio page and no resolved entity id. +// +// So the fact branch returned nothing every time. It did not fail loudly: +// the union just silently reduced to the career_history path, and when that +// path had no row with a resolved organization_entity_id, every founder check +// on the company reported "no founders on record" rather than "we could not +// link them". +// +// Run against the real schema so the SQL is executed, not just pattern- +// matched: an equality that cannot match is exactly what got us here. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { DatabaseSync } from "node:sqlite"; +import { readFileSync, readdirSync } from "node:fs"; +import { join, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = join(dirname(fileURLToPath(import.meta.url)), ".."); +const SRC = readFileSync(join(ROOT, "src/services/diligence/checks/founders.ts"), "utf8"); + +/** The exact SQL the check runs, lifted from the source so they cannot drift. */ +function founderSql() { + const m = SRC.match(/`(SELECT f\.entity_id FROM facts f[\s\S]*?)`/); + assert.ok(m, "the founder lookup in getFoundersOf changed shape"); + return m[1]; +} + +function db() { + const d = new DatabaseSync(":memory:"); + const dir = join(ROOT, "migrations"); + for (const f of readdirSync(dir).filter((x) => x.endsWith(".sql")).sort()) { + d.exec(readFileSync(join(dir, f), "utf8")); + } + return d; +} + +let seq = 0; +function fact(d, { entity, predicate, text = null, entityRef = null }) { + d.prepare( + `INSERT INTO facts (id, entity_id, predicate, value_text, value_entity_id, + source_kind, source, confidence, hash) + VALUES (?, ?, ?, ?, ?, 'scrape', 'test', 0.8, ?)`, + ).run(`f${++seq}`, entity, predicate, text, entityRef, `h${seq}`); +} + +function seedCompany(d, id, name) { + d.prepare("INSERT INTO u_entities (id, kind, display_name) VALUES (?, 'org', ?)").run(id, name); +} +function seedPerson(d, id) { + d.prepare("INSERT INTO u_entities (id, kind, display_name) VALUES (?, 'person', ?)").run(id, id); +} + +const run = (d, companyId) => + d.prepare(founderSql()).all(companyId, companyId).map((r) => r.entity_id).sort(); + +test("a founder recorded by company NAME is found", () => { + const d = db(); + seedCompany(d, "co1", "Acme Robotics"); + seedPerson(d, "p1"); + fact(d, { entity: "p1", predicate: "founder.company_founded", text: " acme robotics " }); + assert.deepEqual(run(d, "co1"), ["p1"], + "this is what the only writer of the predicate actually stores"); +}); + +test("a founder recorded by resolved entity id is still found", () => { + const d = db(); + seedCompany(d, "co1", "Acme Robotics"); + seedPerson(d, "p2"); + fact(d, { entity: "p2", predicate: "founder.company_founded", entityRef: "co1" }); + assert.deepEqual(run(d, "co1"), ["p2"], "the correct shape must keep working"); +}); + +test("a different company's founder is not picked up", () => { + const d = db(); + seedCompany(d, "co1", "Acme Robotics"); + seedCompany(d, "co2", "Globex"); + seedPerson(d, "p3"); + fact(d, { entity: "p3", predicate: "founder.company_founded", text: "Globex" }); + assert.deepEqual(run(d, "co1"), [], "name matching must not widen to every company"); +}); + +test("an empty or whitespace company string matches nothing", () => { + const d = db(); + seedCompany(d, "co1", " "); + seedPerson(d, "p4"); + fact(d, { entity: "p4", predicate: "founder.company_founded", text: "" }); + fact(d, { entity: "p4", predicate: "founder.company_founded", text: " " }); + assert.deepEqual(run(d, "co1"), [], + "TRIM('') = TRIM(' ') would otherwise join every blank-named row to every blank-named company"); +}); + +test("another predicate on the same company is ignored", () => { + const d = db(); + seedCompany(d, "co1", "Acme Robotics"); + seedPerson(d, "p5"); + fact(d, { entity: "p5", predicate: "person.employer", text: "Acme Robotics" }); + assert.deepEqual(run(d, "co1"), []); +}); From 44e7e523db01f970a27b31638827fcc4790ff0f1 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 23:02:53 +0000 Subject: [PATCH 5/7] Per-sector PageRank ran on an empty sector map, every sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit loadPrimarySectors drives the partition computeInfluence uses for per-sector PageRank and the per-sector power-node flags. It read three predicates from `facts`: entity.primary_sector, firm.sector and company.sector. Nothing in the worker writes any of the three. What is written is the PLURAL `firm.sectors`, emitted by the profile workflows as a JSON array in value_json, and `industry`, emitted by the account dual-write as value_text — both of which also become `sector` tags that the summary rebuild materialises into entity_summary.sectors_csv. So the map came back empty on every sweep, every node fell into one unsectored bucket, and sectors_ranked was 0. Indistinguishable from a graph that genuinely has no sector data, which is why nothing ever surfaced it. entity_summary.sectors_csv is now the primary source: it is the materialised one, already deduped and slugged, one row per entity. The facts lookup stays as a fallback for entities whose summary has not been rebuilt, widened to the predicates that are actually written and reading value_json as well as value_text, since the plural forms are arrays. A value_json that is not an array is ignored rather than throwing. The new failure path logs through logError rather than console.warn — the repo's console gate is added-lines-only, so the file's existing console.warn stays and mine would have failed CI. test/edge_quality_sectors.test.mjs lifts both queries out of the source and runs them against the real migrations, because the entire failure was a query that parsed, ran, and returned nothing. Found by scanning every `predicate = '...'` and `predicate IN (...)` read site in the worker against every `predicate: "..."` write site, then keeping only the sites where NO alternative in the list has a writer. That narrowed 43 orphan predicates to 11 read sites. This was the one that silently disabled a whole computation; the rest are either deliberate multi-spelling tolerance, or absences that already degrade visibly (needs_human, caution, a country-centroid geo fallback). Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/package.json | 2 +- apps/worker/src/services/edgeQuality/sweep.ts | 72 ++++++++-- .../worker/test/edge_quality_sectors.test.mjs | 134 ++++++++++++++++++ 3 files changed, 196 insertions(+), 12 deletions(-) create mode 100644 apps/worker/test/edge_quality_sectors.test.mjs diff --git a/apps/worker/package.json b/apps/worker/package.json index 8f1b272a..6050be23 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/services/edgeQuality/sweep.ts b/apps/worker/src/services/edgeQuality/sweep.ts index aa0436dd..b345627c 100644 --- a/apps/worker/src/services/edgeQuality/sweep.ts +++ b/apps/worker/src/services/edgeQuality/sweep.ts @@ -15,6 +15,8 @@ import { collectAllSignals } from "./signals"; import { aggregateSignals } from "./aggregate"; import { computeInfluence, type ScoredEdge } from "./influence"; import { insertFact } from "../../entities/facts"; +import { logError } from "../../db/error_log"; +import { wrapUnknown } from "../../errors"; const EDGE_BATCH = 200; // edges scored per loop iteration const EDGE_TICK_CAP = 5000; // hard ceiling per nightly tick (Task #2 precedent) @@ -298,10 +300,23 @@ async function rebuildEntityInfluence(env: Env): Promise { } /** - * Best-effort sector lookup. Reads the most recent fact with predicate - * `entity.primary_sector` for each id; falls back to `firm.sector` for - * firm entities and `company.sector` for company entities. Empty for - * ids with no sector evidence. + * Best-effort sector lookup, driving the per-sector PageRank partition. + * + * This used to read only `entity.primary_sector`, `firm.sector` and + * `company.sector` from `facts`. No writer in the worker produces any of + * those three — the plural `firm.sectors` is what the profile workflows emit, + * `industry` is what the account dual-write emits, and both land as tags that + * the summary rebuild materialises into `entity_summary.sectors_csv`. So the + * map came back empty on every sweep, every node fell into the same unsectored + * bucket, and `sectors_ranked` was 0: per-sector PageRank and the per-sector + * power-node flags did nothing at all, silently, because an empty map is also + * what a genuinely unsectored graph produces. + * + * `entity_summary.sectors_csv` is now the primary source because it is the + * materialised one — already deduped, already slugged, one row per entity. + * The facts lookup stays as a fallback for entities whose summary has not been + * rebuilt yet, widened to the predicates that are actually written and reading + * `value_json` too, since the plural forms are stored as JSON arrays. */ async function loadPrimarySectors(env: Env, ids: string[]): Promise> { const out = new Map(); @@ -310,16 +325,38 @@ async function loadPrimarySectors(env: Env, ids: string[]): Promise '' + AND entity_id IN (${slice.map(() => "?").join(",")})`, + ).bind(...slice).all<{ entity_id: string; sectors_csv: string | null }>(); + for (const row of r.results ?? []) { + const first = (row.sectors_csv ?? "").split(",").map((x) => x.trim()).find(Boolean); + if (first && !out.has(row.entity_id)) out.set(row.entity_id, first.toLowerCase()); + } + } catch (e) { + await logError(env, { + err: wrapUnknown(e, "db_error", { chunk_start: i, chunk_size: slice.length }), + step: "edge_quality.primary_sector_summary", + }); + } + + const pending = slice.filter((id) => !out.has(id)); + if (!pending.length) continue; + try { + const r = await env.DB.prepare( + `SELECT entity_id, value_text, value_json FROM facts WHERE is_current = 1 - AND predicate IN ('entity.primary_sector','firm.sector','company.sector') - AND entity_id IN (${slice.map(() => "?").join(",")})`, - ).bind(...slice).all<{ entity_id: string; value_text: string | null }>(); + AND predicate IN ('entity.primary_sector','firm.sector','company.sector', + 'firm.sectors','company.sectors','sector','industry', + 'firm.industry') + AND entity_id IN (${pending.map(() => "?").join(",")})`, + ).bind(...pending).all<{ entity_id: string; value_text: string | null; value_json: string | null }>(); for (const row of r.results ?? []) { - if (row.value_text && !out.has(row.entity_id)) { - out.set(row.entity_id, row.value_text.toLowerCase()); - } + if (out.has(row.entity_id)) continue; + const v = row.value_text?.trim() || firstOfJsonArray(row.value_json); + if (v) out.set(row.entity_id, v.toLowerCase()); } } catch (e) { console.warn("primary sector lookup chunk failed", (e as Error).message); @@ -328,6 +365,19 @@ async function loadPrimarySectors(env: Env, ids: string[]): Promise x.endsWith(".sql")).sort()) { + d.exec(readFileSync(join(dir, f), "utf8")); + } + return d; +} + +/** Lift both queries out of the source so test and implementation cannot drift. */ +function queries() { + const summary = SRC.match(/`(SELECT entity_id, sectors_csv[\s\S]*?)`/); + const facts = SRC.match(/`(SELECT entity_id, value_text, value_json[\s\S]*?)`/); + assert.ok(summary, "the entity_summary sector query changed shape"); + assert.ok(facts, "the facts sector fallback changed shape"); + // One placeholder per bound id; the tests bind exactly one. + return { + summary: summary[1].replace(/\$\{[^}]*\}/, "?"), + facts: facts[1].replace(/\$\{[^}]*\}/, "?"), + }; +} + +let seq = 0; +const ent = (d, id) => d.prepare("INSERT INTO u_entities (id, kind, display_name) VALUES (?, 'org', ?)").run(id, id); +const fact = (d, id, predicate, { text = null, json = null } = {}) => + d.prepare( + `INSERT INTO facts (id, entity_id, predicate, value_text, value_json, + source_kind, source, confidence, hash) + VALUES (?, ?, ?, ?, ?, 'scrape', 'test', 0.8, ?)`, + ).run(`f${++seq}`, id, predicate, text, json, `h${seq}`); + +/** Mirrors loadPrimarySectors: summary first, facts for what is left. */ +function resolve(d, id) { + const q = queries(); + const s = d.prepare(q.summary).all(id); + for (const row of s) { + const first = (row.sectors_csv ?? "").split(",").map((x) => x.trim()).find(Boolean); + if (first) return first.toLowerCase(); + } + for (const row of d.prepare(q.facts).all(id)) { + const v = row.value_text?.trim() || firstOfJson(row.value_json); + if (v) return v.toLowerCase(); + } + return null; +} +function firstOfJson(raw) { + if (!raw) return null; + try { + const p = JSON.parse(raw); + if (!Array.isArray(p)) return null; + for (const x of p) if (typeof x === "string" && x.trim()) return x.trim(); + } catch { /* not JSON */ } + return null; +} + +test("entity_summary.sectors_csv resolves — it is the materialised source", () => { + const d = db(); + ent(d, "e1"); + d.prepare("INSERT INTO entity_summary (entity_id, kind, sectors_csv) VALUES (?, 'org', ?)") + .run("e1", "Fintech,SaaS"); + assert.equal(resolve(d, "e1"), "fintech"); +}); + +test("the PLURAL firm.sectors JSON array resolves — this is what writers emit", () => { + const d = db(); + ent(d, "e2"); + fact(d, "e2", "firm.sectors", { json: JSON.stringify(["Climate", "Energy"]) }); + assert.equal(resolve(d, "e2"), "climate", + "firm.sectors is written as a JSON array by the profile workflows; the old query read only value_text on the singular name"); +}); + +test("`industry` resolves — this is what the account dual-write emits", () => { + const d = db(); + ent(d, "e3"); + fact(d, "e3", "industry", { text: "Logistics" }); + assert.equal(resolve(d, "e3"), "logistics"); +}); + +test("the singular predicates still resolve if anything ever writes one", () => { + const d = db(); + ent(d, "e4"); + fact(d, "e4", "entity.primary_sector", { text: "Healthcare" }); + assert.equal(resolve(d, "e4"), "healthcare"); +}); + +test("an entity with no sector evidence stays unsectored", () => { + const d = db(); + ent(d, "e5"); + fact(d, "e5", "person.title", { text: "Partner" }); + assert.equal(resolve(d, "e5"), null, "an unrelated predicate must not become a sector"); +}); + +test("a summary row with an empty sectors_csv falls through to facts", () => { + const d = db(); + ent(d, "e6"); + d.prepare("INSERT INTO entity_summary (entity_id, kind, sectors_csv) VALUES (?, 'org', ?)").run("e6", ""); + fact(d, "e6", "firm.sectors", { json: JSON.stringify(["Biotech"]) }); + assert.equal(resolve(d, "e6"), "biotech", + "an entity whose summary has been rebuilt with no tags must still use its facts"); +}); + +test("a non-array value_json does not crash the fallback", () => { + const d = db(); + ent(d, "e7"); + fact(d, "e7", "firm.sectors", { json: '{"not":"an array"}' }); + assert.equal(resolve(d, "e7"), null); +}); From 7ee85015dc1cff81c1ca88549b36c807489bf522 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 23:06:58 +0000 Subject: [PATCH 6/7] Filtering a comp panel by sector returned an empty panel, always MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The valuation module has four sector lookups, each asking `facts` for `company.sector`, `firm.sector` or `sector`. No writer in the worker produces any of the three. What exists is the plural `firm.sectors` (a JSON array in value_json, from the profile workflows), `industry` (value_text, from the account dual-write), and entity_summary.sectors_csv (the materialised slug list the summary rebuild derives from sector tags). In the comp panel screen a sector miss does not widen the result — it `continue`s past the candidate. So an operator who filtered by sector got an empty panel and the reasonable conclusion that nothing was comparable, rather than any indication that the filter could not match. The private- member discovery query had the same predicates in a JOIN, so it returned nothing for every sector. getCompanySector in impliedValuation returned null for every company, so its sector-matched panel fallback never fired. entities/sector.ts is now the one place that knows the storage shapes: entityHasSector for a per-entity screen, entityPrimarySector for "what is this company's sector", and a literal EXISTS fragment for the query that needs it inline. The edge-quality sweep's copy of the JSON-array helper is gone in favour of the shared one. Two details worth keeping: - sectors_csv is matched with comma-delimited instr, not a substring test. A bare instr matches "fin" inside "fintech" and would widen every filter it was meant to narrow. monitoring/smart.ts already uses this technique against the same column. - The fragment is a literal constant rather than a template. The only variable part would have been the column name, and the repo's SQL gate forbids interpolating into a statement; a caller needing a different column inlines its own copy instead of concatenating one. Deliberately left alone: `company.business_model` in the same screen also has no writer, but unlike sector there is no alternative spelling anywhere in the schema to converge on. Widening it would mean inventing a source. test/sector_resolution.test.mjs lifts the SQL out of the module and runs it against the real migrations, covering each storage shape, the delimiter, superseded facts, and an empty filter argument — which must match nothing rather than everything. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/package.json | 2 +- apps/worker/src/entities/sector.ts | 135 ++++++++++++++++++ apps/worker/src/services/edgeQuality/sweep.ts | 13 +- .../src/services/valuation/compPanel.ts | 22 +-- .../services/valuation/impliedValuation.ts | 11 +- apps/worker/test/sector_resolution.test.mjs | 129 +++++++++++++++++ 6 files changed, 284 insertions(+), 28 deletions(-) create mode 100644 apps/worker/src/entities/sector.ts create mode 100644 apps/worker/test/sector_resolution.test.mjs diff --git a/apps/worker/package.json b/apps/worker/package.json index 6050be23..3c8e99e8 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs test/sector_resolution.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/entities/sector.ts b/apps/worker/src/entities/sector.ts new file mode 100644 index 00000000..db079eee --- /dev/null +++ b/apps/worker/src/entities/sector.ts @@ -0,0 +1,135 @@ +// One place that knows how a sector is actually stored. +// +// Four call sites in the valuation module and one in the edge-quality sweep +// each asked `facts` for `company.sector`, `firm.sector` or `sector`. No +// writer in the worker produces any of those. What is written is: +// +// * `firm.sectors` / `company.sectors` — a JSON ARRAY in value_json, +// emitted by the profile workflows (crawler/profileWorkflows/_commonSchemas). +// * `industry` — value_text, emitted by the account dual-write. +// * `entity_summary.sectors_csv` — the materialised, deduped, slugged list +// the summary rebuild derives from `sector` tags. +// +// Reading only the three singular text predicates meant every sector lookup +// returned nothing. In the comp panel that is worse than empty: the screen +// `continue`s past any candidate whose sector does not match, so an operator +// filtering by sector got an empty panel and the reasonable conclusion that +// there were no comparable companies. +// +// These helpers exist so the next reader does not have to rediscover which of +// the six spellings is the live one. + +import type { Env } from "../types"; + +/** Predicates whose value lives in value_text. */ +export const SECTOR_TEXT_PREDICATES = [ + "entity.primary_sector", "company.sector", "firm.sector", + "sector", "industry", "firm.industry", +] as const; + +/** Predicates whose value is a JSON array in value_json. */ +export const SECTOR_ARRAY_PREDICATES = [ + "firm.sectors", "company.sectors", "sectors", +] as const; + +/** + * True when `entityId` carries `sector` under any storage shape. + * + * Written as one statement so a per-row screen costs one round trip. The + * summary arm uses comma-delimited `instr` — the same technique + * monitoring/smart.ts uses against the same column — because `sectors_csv` + * is a bare join of slugs, so a substring test alone would match "fin" inside + * "fintech". + */ +export async function entityHasSector(env: Env, entityId: string, sector: string): Promise { + const wanted = sector.trim(); + if (!entityId || !wanted) return false; + const r = await env.DB.prepare( + `SELECT 1 AS hit FROM entity_summary s + WHERE s.entity_id = ? + AND s.sectors_csv IS NOT NULL + AND instr(',' || lower(s.sectors_csv) || ',', ',' || lower(?) || ',') > 0 + UNION ALL + SELECT 1 AS hit FROM facts f + WHERE f.entity_id = ? + AND f.is_current = 1 + AND ( + (f.predicate IN ('entity.primary_sector','company.sector','firm.sector','sector','industry','firm.industry') + AND lower(trim(f.value_text)) = lower(?)) + OR (f.predicate IN ('firm.sectors','company.sectors','sectors') + AND f.value_json IS NOT NULL + AND EXISTS (SELECT 1 FROM json_each(f.value_json) je + WHERE lower(trim(je.value)) = lower(?))) + ) + LIMIT 1`, + ).bind(entityId, wanted, entityId, wanted, wanted).first<{ hit: number }>(); + return Boolean(r); +} + +/** + * The entity's primary sector, lower-cased, or null when there is no evidence. + * Prefers `entity_summary` because it is the materialised list; falls back to + * facts for entities whose summary has not been rebuilt yet. + */ +export async function entityPrimarySector(env: Env, entityId: string): Promise { + if (!entityId) return null; + const sum = await env.DB.prepare( + `SELECT sectors_csv FROM entity_summary WHERE entity_id = ?`, + ).bind(entityId).first<{ sectors_csv: string | null }>(); + const fromSummary = (sum?.sectors_csv ?? "").split(",").map((x) => x.trim()).find(Boolean); + if (fromSummary) return fromSummary.toLowerCase(); + + const f = await env.DB.prepare( + `SELECT value_text, value_json FROM facts + WHERE entity_id = ? + AND is_current = 1 + AND predicate IN ('entity.primary_sector','company.sector','firm.sector','sector', + 'industry','firm.industry','firm.sectors','company.sectors','sectors') + ORDER BY observed_at DESC + LIMIT 1`, + ).bind(entityId).first<{ value_text: string | null; value_json: string | null }>(); + const text = f?.value_text?.trim(); + if (text) return text.toLowerCase(); + return firstOfJsonArray(f?.value_json ?? null); +} + +/** First non-empty string in a JSON array column, lower-cased; null otherwise. */ +export function firstOfJsonArray(raw: string | null): string | null { + if (!raw) return null; + try { + const parsed = JSON.parse(raw) as unknown; + if (!Array.isArray(parsed)) return null; + for (const x of parsed) { + if (typeof x === "string" && x.trim()) return x.trim().toLowerCase(); + } + } catch { /* not JSON — nothing to take */ } + return null; +} + +/** + * `EXISTS (...)` fragment matching a sector against a column already in scope. + * + * A literal, not a template: the repo's SQL gate forbids interpolating into a + * statement, and the only variable part here would have been the column name. + * Callers that need a different column inline their own copy rather than + * building one by concatenation. + * + * Binds, in order: sector, sector, sector. + */ +export const SECTOR_MATCHES_COMPANY_ENTITY_SQL = `( + EXISTS (SELECT 1 FROM entity_summary s + WHERE s.entity_id = vm.company_entity_id + AND s.sectors_csv IS NOT NULL + AND instr(',' || lower(s.sectors_csv) || ',', ',' || lower(?) || ',') > 0) + OR EXISTS (SELECT 1 FROM facts f + WHERE f.entity_id = vm.company_entity_id + AND f.is_current = 1 + AND ( + (f.predicate IN ('entity.primary_sector','company.sector','firm.sector','sector','industry','firm.industry') + AND lower(trim(f.value_text)) = lower(?)) + OR (f.predicate IN ('firm.sectors','company.sectors','sectors') + AND f.value_json IS NOT NULL + AND EXISTS (SELECT 1 FROM json_each(f.value_json) je + WHERE lower(trim(je.value)) = lower(?))) + )) +)`; diff --git a/apps/worker/src/services/edgeQuality/sweep.ts b/apps/worker/src/services/edgeQuality/sweep.ts index b345627c..b19e5a4c 100644 --- a/apps/worker/src/services/edgeQuality/sweep.ts +++ b/apps/worker/src/services/edgeQuality/sweep.ts @@ -15,6 +15,7 @@ import { collectAllSignals } from "./signals"; import { aggregateSignals } from "./aggregate"; import { computeInfluence, type ScoredEdge } from "./influence"; import { insertFact } from "../../entities/facts"; +import { firstOfJsonArray } from "../../entities/sector"; import { logError } from "../../db/error_log"; import { wrapUnknown } from "../../errors"; @@ -365,18 +366,6 @@ async function loadPrimarySectors(env: Env, ids: string[]): Promise(); + ).bind(criteria.sector, criteria.sector, criteria.sector) + .all<{ company_entity_id: string; display_name: string }>(); for (const r of (privRows.results ?? [])) { priv.push({ company_entity_id: r.company_entity_id, company_name: r.display_name, diff --git a/apps/worker/src/services/valuation/impliedValuation.ts b/apps/worker/src/services/valuation/impliedValuation.ts index 09d54cdb..844a6ac5 100644 --- a/apps/worker/src/services/valuation/impliedValuation.ts +++ b/apps/worker/src/services/valuation/impliedValuation.ts @@ -19,6 +19,7 @@ // selection can be extended to prefer panels whose stage matches the // target's most recent funding round. import type { Env } from "../../types"; +import { entityPrimarySector } from "../../entities/sector"; import type { ImpliedValuationRange } from "./types"; function percentile(sorted: number[], p: number): number { @@ -27,13 +28,11 @@ function percentile(sorted: number[], p: number): number { return sorted[idx]; } +// Delegates to entities/sector.ts. The three predicates this used to read +// have no writer anywhere in the worker, so it returned null for every +// company and the sector-matched panel fallback below never fired. async function getCompanySector(env: Env, entityId: string): Promise { - const r = await env.DB.prepare( - `SELECT value_text FROM facts - WHERE entity_id = ? AND predicate IN ('company.sector','firm.sector','sector') - AND is_current = 1 LIMIT 1`, - ).bind(entityId).first<{ value_text: string }>(); - return r?.value_text ?? null; + return entityPrimarySector(env, entityId); } async function pickPanelForCompany(env: Env, entityId: string): Promise<{ id: string; name: string; criteria_json: string } | null> { diff --git a/apps/worker/test/sector_resolution.test.mjs b/apps/worker/test/sector_resolution.test.mjs new file mode 100644 index 00000000..69f9406a --- /dev/null +++ b/apps/worker/test/sector_resolution.test.mjs @@ -0,0 +1,129 @@ +// Every sector lookup in the platform read predicates nothing writes. +// +// Four call sites in the valuation module and one in the edge-quality sweep +// asked `facts` for `company.sector`, `firm.sector` or `sector`. No writer in +// the worker produces any of the three. What exists is the PLURAL +// `firm.sectors` (a JSON array in value_json, from the profile workflows), +// `industry` (value_text, from the account dual-write), and +// entity_summary.sectors_csv (the materialised slug list the summary rebuild +// derives from `sector` tags). +// +// In the comp panel that is worse than returning everything: a sector miss +// `continue`s past the candidate, so an operator filtering by sector got an +// EMPTY panel and the reasonable conclusion that nothing was comparable. +// +// entities/sector.ts is now the single place that knows the storage shapes. +// Its SQL is lifted out of the source and executed against the real +// migrations, because a query that parses, runs and matches nothing is +// exactly what this is about. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { DatabaseSync } from "node:sqlite"; +import { readFileSync, readdirSync } from "node:fs"; +import { join, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = join(dirname(fileURLToPath(import.meta.url)), ".."); +const SRC = readFileSync(join(ROOT, "src/entities/sector.ts"), "utf8"); + +function db() { + const d = new DatabaseSync(":memory:"); + const dir = join(ROOT, "migrations"); + for (const f of readdirSync(dir).filter((x) => x.endsWith(".sql")).sort()) { + d.exec(readFileSync(join(dir, f), "utf8")); + } + return d; +} + +function hasSectorSql() { + const m = SRC.match(/`(SELECT 1 AS hit FROM entity_summary s[\s\S]*?)`/); + assert.ok(m, "entityHasSector's statement changed shape"); + return m[1]; +} + +let seq = 0; +const org = (d, id, name = id) => + d.prepare("INSERT INTO u_entities (id, kind, display_name) VALUES (?, 'org', ?)").run(id, name); +const summary = (d, id, csv) => + d.prepare("INSERT INTO entity_summary (entity_id, kind, sectors_csv) VALUES (?, 'org', ?)").run(id, csv); +const fact = (d, id, predicate, { text = null, json = null } = {}) => + d.prepare( + `INSERT INTO facts (id, entity_id, predicate, value_text, value_json, + source_kind, source, confidence, hash) + VALUES (?, ?, ?, ?, ?, 'scrape', 'test', 0.8, ?)`, + ).run(`f${++seq}`, id, predicate, text, json, `h${seq}`); + +const has = (d, id, sector) => + d.prepare(hasSectorSql()).all(id, sector, id, sector, sector).length > 0; + +test("the module still exports what the call sites import", () => { + for (const name of ["entityHasSector", "entityPrimarySector", "SECTOR_MATCHES_COMPANY_ENTITY_SQL"]) { + assert.match(SRC, new RegExp(`export (?:async function|function|const) ${name}\\b`), `${name} missing`); + } +}); + +test("matches the materialised entity_summary.sectors_csv", () => { + const d = db(); org(d, "c1"); summary(d, "c1", "fintech,saas"); + assert.equal(has(d, "c1", "SaaS"), true, "match must be case-insensitive"); + assert.equal(has(d, "c1", "biotech"), false); +}); + +test("sectors_csv matching is delimited, not a substring test", () => { + // "fin" inside "fintech" must not match, or every sector filter widens. + const d = db(); org(d, "c2"); summary(d, "c2", "fintech"); + assert.equal(has(d, "c2", "fin"), false); + assert.equal(has(d, "c2", "tech"), false); + assert.equal(has(d, "c2", "fintech"), true); +}); + +test("matches the PLURAL firm.sectors JSON array — what writers emit", () => { + const d = db(); org(d, "c3"); + fact(d, "c3", "firm.sectors", { json: JSON.stringify(["Climate", " Energy "]) }); + assert.equal(has(d, "c3", "climate"), true); + assert.equal(has(d, "c3", "energy"), true, "array members are trimmed before comparison"); + assert.equal(has(d, "c3", "fintech"), false); +}); + +test("matches `industry` — what the account dual-write emits", () => { + const d = db(); org(d, "c4"); + fact(d, "c4", "industry", { text: " Logistics " }); + assert.equal(has(d, "c4", "logistics"), true); +}); + +test("the singular predicates still match if anything ever writes one", () => { + const d = db(); org(d, "c5"); + fact(d, "c5", "company.sector", { text: "Healthcare" }); + assert.equal(has(d, "c5", "healthcare"), true); +}); + +test("a superseded fact does not match", () => { + const d = db(); org(d, "c6"); + fact(d, "c6", "industry", { text: "Logistics" }); + d.prepare("UPDATE facts SET is_current = 0 WHERE entity_id = ?").run("c6"); + assert.equal(has(d, "c6", "logistics"), false); +}); + +test("an unrelated predicate is not treated as a sector", () => { + const d = db(); org(d, "c7"); + fact(d, "c7", "person.title", { text: "Fintech" }); + assert.equal(has(d, "c7", "fintech"), false); +}); + +test("an empty sector argument matches nothing", () => { + const d = db(); org(d, "c8"); summary(d, "c8", "fintech"); + assert.equal(has(d, "c8", ""), false, + "an empty filter must not silently match every entity"); +}); + +test("the valuation call sites use the helper rather than their own SQL", () => { + const comp = readFileSync(join(ROOT, "src/services/valuation/compPanel.ts"), "utf8"); + const impl = readFileSync(join(ROOT, "src/services/valuation/impliedValuation.ts"), "utf8"); + assert.match(comp, /entityHasSector\(env, r\.company_entity_id, criteria\.sector\)/); + assert.match(comp, /SECTOR_MATCHES_COMPANY_ENTITY_SQL/); + assert.match(impl, /entityPrimarySector\(env, entityId\)/); + for (const [name, s] of [["compPanel", comp], ["impliedValuation", impl]]) { + assert.ok(!/predicate IN \('company\.sector','firm\.sector','sector'\)/.test(s), + `${name} still has its own writer-less sector query`); + } +}); From 418d7a7c02f0043e03be417373fffe0a900fc453 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 6 Sep 2026 23:12:09 +0000 Subject: [PATCH 7/7] The weekly firm team snapshot picked zero firms, forever MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit runWeeklySnapshotSweep is the only producer for firm_team_snapshots, which the spinout detector (movements/spinout.ts) reads to notice partners leaving a firm. Its eligibility query required BOTH: * a current `firm.team_url` fact — a predicate no writer in the worker produces, and * entity_roles.role = 'investor_firm' — assigned only by deals/investorResolver, while every firm that got its role through the dual-write carries 'firm'. Either condition alone was fatal. The sweep returned picked:0, which is also exactly what "every firm is already up to date" looks like, so nothing ever surfaced it — and the whole spinout-detection chain behind it had no input. firm_people.source_url is the page scraper/pipeline.ts actually parsed a firm's people from, and the only populated team-page URL in the schema. The query now unions it in as a second-preference candidate behind the explicit fact, so an operator override still wins where one exists, and ROW_NUMBER collapses the two sources to one row per firm. The join casts firms.id to TEXT to reach entity_legacy_map, matching how portfolioFromFirmSite.ts and roleInference.ts already bridge that gap. The role filter is widened to the set the sibling detector uses. Two files in the same feature reading the same graph through different role sets was itself part of the bug. test/team_snapshot_eligibility.test.mjs lifts the query out of the source and runs it against the real migrations: the scraped-page path, both role spellings, override precedence, the 7-day window, a non-firm entity, and a firm with no team page anywhere. Checked and deliberately not changed: `firm.companies_house_number` also has no writer, but it is a fallback behind an explicit uk_company_number and the whole branch is gated on COMPANIES_HOUSE_API_KEY, so it degrades cleanly. Correction to an earlier suspicion: admin.ts's backfill-coverage counters join entity_legacy_map without a CAST, and firms.id/companies.id are INTEGER against a TEXT legacy_id. I expected that never to match. It does — SQLite applies numeric affinity to the TEXT operand — verified against the same engine before touching it. Those counters are correct; no change. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EgworUXKA6pZzEcUUJXCmA --- apps/worker/package.json | 2 +- .../worker/src/services/movements/snapshot.ts | 43 +- apps/worker/test-dist-q/ai/budget.js | 46 ++ apps/worker/test-dist-q/ai/cache.js | 33 + apps/worker/test-dist-q/ai/extract.js | 335 +++++++++ apps/worker/test-dist-q/analytics/events.js | 79 ++ apps/worker/test-dist-q/db/error_log.js | 125 ++++ apps/worker/test-dist-q/entities/channels.js | 61 ++ apps/worker/test-dist-q/entities/dualwrite.js | 375 ++++++++++ apps/worker/test-dist-q/entities/facts.js | 282 ++++++++ apps/worker/test-dist-q/entities/garbage.js | 652 +++++++++++++++++ apps/worker/test-dist-q/entities/model.js | 6 + apps/worker/test-dist-q/entities/normalize.js | 120 +++ .../entities/profile-predicates.js | 218 ++++++ .../test-dist-q/entities/profile-shapes.js | 4 + apps/worker/test-dist-q/entities/profile.js | 681 ++++++++++++++++++ apps/worker/test-dist-q/entities/query.js | 154 ++++ apps/worker/test-dist-q/entities/roles.js | 120 +++ apps/worker/test-dist-q/entities/summary.js | 121 ++++ .../test-dist-q/entities/summaryQueue.js | 26 + apps/worker/test-dist-q/entities/tags.js | 37 + apps/worker/test-dist-q/errors.js | 328 +++++++++ apps/worker/test-dist-q/personas/repo.js | 403 +++++++++++ apps/worker/test-dist-q/personas/score.js | 281 ++++++++ .../test-dist-q/scraper/firms_upsert.js | 267 +++++++ apps/worker/test-dist-q/scraper/normalize.js | 104 +++ .../scraper/parsers/firmlists/types.js | 1 + apps/worker/test-dist-q/scraper/rateLimit.js | 53 ++ .../services/personaMatchTrigger.js | 59 ++ .../test-dist-q/services/personaMatching.js | 484 +++++++++++++ .../services/personaMatchingScorers.js | 366 ++++++++++ .../services/personas/kinds/_generic.js | 45 ++ .../personas/kinds/academic_researcher.js | 7 + .../personas/kinds/account_company.js | 17 + .../services/personas/kinds/acquirer.js | 7 + .../personas/kinds/angel_individual.js | 7 + .../services/personas/kinds/beta_tester.js | 7 + .../services/personas/kinds/buyer_person.js | 37 + .../personas/kinds/channel_partner.js | 7 + .../personas/kinds/co_founder_match.js | 7 + .../services/personas/kinds/competitor.js | 7 + .../services/personas/kinds/design_partner.js | 7 + .../personas/kinds/engineering_hire.js | 7 + .../services/personas/kinds/executive_hire.js | 7 + .../services/personas/kinds/founder.js | 68 ++ .../personas/kinds/fractional_executive.js | 7 + .../kinds/government_grant_officer.js | 7 + .../services/personas/kinds/index.js | 71 ++ .../personas/kinds/integration_partner.js | 7 + .../services/personas/kinds/investor_firm.js | 64 ++ .../personas/kinds/investor_person.js | 14 + .../personas/kinds/journalist_analyst.js | 7 + .../personas/kinds/limited_partner.js | 7 + .../services/personas/kinds/policy_advisor.js | 7 + .../services/personas/kinds/regulator.js | 7 + .../personas/kinds/service_provider.js | 7 + .../services/personas/kinds/taxonomy.js | 86 +++ .../services/personas/kinds/thought_leader.js | 7 + .../personas/kinds/venture_partner.js | 50 ++ .../services/relationships/_safeQuery.js | 18 + .../services/relationships/baselines.js | 42 ++ .../extractors/advisorFromBio.js | 38 + .../extractors/boardSeatFromFilings.js | 78 ++ .../extractors/coAuthorFromPublications.js | 35 + .../extractors/coInvestorFromDeals.js | 42 ++ .../extractors/colleagueOverlap.js | 45 ++ .../extractors/educationFromBio.js | 36 + .../employmentHistoryFromLinkedIn.js | 44 ++ .../extractors/familyFromPublicSources.js | 40 + .../extractors/investedInFromDeals.js | 49 ++ .../extractors/mentionFromNews.js | 40 + .../extractors/portfolioFromFirmSite.js | 47 ++ .../relationships/extractors/schoolWith.js | 35 + .../extractors/worksAtFromTitle.js | 46 ++ .../services/relationships/orchestrator.js | 116 +++ .../services/relationships/persist.js | 92 +++ .../services/relationships/resolve.js | 58 ++ .../services/relationships/types.js | 2 + .../test-dist-q/services/roleInference.js | 123 ++++ .../test-dist-q/services/secEdgar/xref.js | 128 ++++ apps/worker/test-dist-q/types.js | 1 + .../test/team_snapshot_eligibility.test.mjs | 128 ++++ 82 files changed, 7726 insertions(+), 8 deletions(-) create mode 100644 apps/worker/test-dist-q/ai/budget.js create mode 100644 apps/worker/test-dist-q/ai/cache.js create mode 100644 apps/worker/test-dist-q/ai/extract.js create mode 100644 apps/worker/test-dist-q/analytics/events.js create mode 100644 apps/worker/test-dist-q/db/error_log.js create mode 100644 apps/worker/test-dist-q/entities/channels.js create mode 100644 apps/worker/test-dist-q/entities/dualwrite.js create mode 100644 apps/worker/test-dist-q/entities/facts.js create mode 100644 apps/worker/test-dist-q/entities/garbage.js create mode 100644 apps/worker/test-dist-q/entities/model.js create mode 100644 apps/worker/test-dist-q/entities/normalize.js create mode 100644 apps/worker/test-dist-q/entities/profile-predicates.js create mode 100644 apps/worker/test-dist-q/entities/profile-shapes.js create mode 100644 apps/worker/test-dist-q/entities/profile.js create mode 100644 apps/worker/test-dist-q/entities/query.js create mode 100644 apps/worker/test-dist-q/entities/roles.js create mode 100644 apps/worker/test-dist-q/entities/summary.js create mode 100644 apps/worker/test-dist-q/entities/summaryQueue.js create mode 100644 apps/worker/test-dist-q/entities/tags.js create mode 100644 apps/worker/test-dist-q/errors.js create mode 100644 apps/worker/test-dist-q/personas/repo.js create mode 100644 apps/worker/test-dist-q/personas/score.js create mode 100644 apps/worker/test-dist-q/scraper/firms_upsert.js create mode 100644 apps/worker/test-dist-q/scraper/normalize.js create mode 100644 apps/worker/test-dist-q/scraper/parsers/firmlists/types.js create mode 100644 apps/worker/test-dist-q/scraper/rateLimit.js create mode 100644 apps/worker/test-dist-q/services/personaMatchTrigger.js create mode 100644 apps/worker/test-dist-q/services/personaMatching.js create mode 100644 apps/worker/test-dist-q/services/personaMatchingScorers.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/_generic.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/account_company.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/acquirer.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/angel_individual.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/beta_tester.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/buyer_person.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/channel_partner.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/competitor.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/design_partner.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/executive_hire.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/founder.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/index.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/integration_partner.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/investor_firm.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/investor_person.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/limited_partner.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/regulator.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/service_provider.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/taxonomy.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/thought_leader.js create mode 100644 apps/worker/test-dist-q/services/personas/kinds/venture_partner.js create mode 100644 apps/worker/test-dist-q/services/relationships/_safeQuery.js create mode 100644 apps/worker/test-dist-q/services/relationships/baselines.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/advisorFromBio.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/boardSeatFromFilings.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/coAuthorFromPublications.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/coInvestorFromDeals.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/colleagueOverlap.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/educationFromBio.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/employmentHistoryFromLinkedIn.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/familyFromPublicSources.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/investedInFromDeals.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/mentionFromNews.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/portfolioFromFirmSite.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/schoolWith.js create mode 100644 apps/worker/test-dist-q/services/relationships/extractors/worksAtFromTitle.js create mode 100644 apps/worker/test-dist-q/services/relationships/orchestrator.js create mode 100644 apps/worker/test-dist-q/services/relationships/persist.js create mode 100644 apps/worker/test-dist-q/services/relationships/resolve.js create mode 100644 apps/worker/test-dist-q/services/relationships/types.js create mode 100644 apps/worker/test-dist-q/services/roleInference.js create mode 100644 apps/worker/test-dist-q/services/secEdgar/xref.js create mode 100644 apps/worker/test-dist-q/types.js create mode 100644 apps/worker/test/team_snapshot_eligibility.test.mjs diff --git a/apps/worker/package.json b/apps/worker/package.json index 3c8e99e8..633160e5 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs test/sector_resolution.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs test/sector_resolution.test.mjs test/team_snapshot_eligibility.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/services/movements/snapshot.ts b/apps/worker/src/services/movements/snapshot.ts index 44e54b76..79df8ef5 100644 --- a/apps/worker/src/services/movements/snapshot.ts +++ b/apps/worker/src/services/movements/snapshot.ts @@ -225,19 +225,48 @@ export async function runWeeklySnapshotSweep(env: Env, limit = 25): Promise<{ // one current `firm.team_url` fact, pick the most recently observed // (then created) row so the crawl target is stable across runs. const rows = await env.DB.prepare( - `WITH picked AS ( + `WITH candidates AS ( + -- Preference 1: an explicit firm.team_url fact. Nothing in the worker + -- writes this predicate, so on its own the sweep picked zero firms + -- forever — and reported picked:0, which is also what "every firm is + -- up to date" looks like. Kept first because it is the right answer + -- when something does set it, e.g. an operator override. SELECT f.entity_id AS firm_entity_id, f.value_text AS team_url, - ROW_NUMBER() OVER ( - PARTITION BY f.entity_id - ORDER BY f.observed_at DESC, f.created_at DESC, f.id ASC - ) AS rn + 1 AS pref, f.observed_at AS ts_a, f.created_at AS ts_b, f.id AS tie FROM facts f - JOIN entity_roles r ON r.entity_id = f.entity_id - AND r.role = 'investor_firm' WHERE f.predicate = 'firm.team_url' AND f.is_current = 1 AND f.value_text IS NOT NULL AND f.value_text <> '' + UNION ALL + -- Preference 2: the page the firm's people were actually scraped from. + -- scraper/pipeline.ts stamps firm_people.source_url with the team page + -- it parsed, which is exactly the URL this sweep wants to re-fetch. It + -- is the only populated source of a firm team URL in the schema. + -- CAST matches how portfolioFromFirmSite.ts and roleInference.ts join + -- firms into entity_legacy_map: firms.id is INTEGER, legacy_id is TEXT. + SELECT m.entity_id AS firm_entity_id, fp.source_url AS team_url, + 2 AS pref, fp.created_at AS ts_a, fp.created_at AS ts_b, CAST(fp.id AS TEXT) AS tie + FROM firm_people fp + JOIN entity_legacy_map m ON m.legacy_table = 'firms' + AND m.legacy_id = CAST(fp.firm_id AS TEXT) + WHERE fp.source_url IS NOT NULL + AND fp.source_url <> '' + ), + picked AS ( + SELECT c.firm_entity_id, c.team_url, + ROW_NUMBER() OVER ( + PARTITION BY c.firm_entity_id + ORDER BY c.pref ASC, c.ts_a DESC, c.ts_b DESC, c.tie ASC + ) AS rn + FROM candidates c + -- Widened from role = 'investor_firm', which only deals/investorResolver + -- assigns. Every firm that got its role from the dual-write carries + -- 'firm', so the old filter excluded the bulk of the table. This is + -- the same set the sibling detector (movements/spinout.ts) uses, and + -- the two reading the same graph differently was itself the bug. + JOIN entity_roles r ON r.entity_id = c.firm_entity_id + AND r.role IN ('investor_firm','firm','fund','vc','gp','investor') ), dated AS ( SELECT p.firm_entity_id, p.team_url, diff --git a/apps/worker/test-dist-q/ai/budget.js b/apps/worker/test-dist-q/ai/budget.js new file mode 100644 index 00000000..3cfcadb5 --- /dev/null +++ b/apps/worker/test-dist-q/ai/budget.js @@ -0,0 +1,46 @@ +// AI / Vectorize daily budget caps (Task #25 step 10). +// +// Reads AI_DAILY_NEURONS_CAP and VECTORIZE_DAILY_QUERIES_CAP from env vars. +// Polls the D1 ai_cost_daily roll-up (cached 60s in KV) for the running +// total; refuses calls past the cap so a runaway loop can't drain the +// account. /api/scrapers/health surfaces burn-down for the dashboard. +const KV_KEY = "ai-budget:today"; +const CACHE_TTL = 60; +export async function getBurn(env) { + const cached = await env.SCRAPE_CACHE?.get(KV_KEY); + if (cached) { + try { + return JSON.parse(cached); + } + catch { /* fall through */ } + } + const day = new Date().toISOString().slice(0, 10); + const totals = await env.DB.prepare(`SELECT + SUM(neurons) AS neurons, + SUM(cost_usd) AS cost, + SUM(CASE WHEN purpose LIKE 'vectorize_%' THEN calls ELSE 0 END) AS vec_calls + FROM ai_cost_daily WHERE day = ?`).bind(day).first().catch(() => null); + const snap = { + day, + neurons_used: Number(totals?.neurons ?? 0), + neurons_cap: Number(env.AI_DAILY_NEURONS_CAP ?? "0") || 0, + vectorize_used: Number(totals?.vec_calls ?? 0), + vectorize_cap: Number(env.VECTORIZE_DAILY_QUERIES_CAP ?? "0") || 0, + cost_usd: Number(totals?.cost ?? 0), + }; + try { + await env.SCRAPE_CACHE?.put(KV_KEY, JSON.stringify(snap), { expirationTtl: CACHE_TTL }); + } + catch { /* best-effort */ } + return snap; +} +export async function assertBudget(env, kind) { + const snap = await getBurn(env); + if (kind === "ai" && snap.neurons_cap > 0 && snap.neurons_used >= snap.neurons_cap) { + return { ok: false, reason: `neurons_cap_reached:${snap.neurons_used}/${snap.neurons_cap}` }; + } + if (kind === "vectorize" && snap.vectorize_cap > 0 && snap.vectorize_used >= snap.vectorize_cap) { + return { ok: false, reason: `vectorize_cap_reached:${snap.vectorize_used}/${snap.vectorize_cap}` }; + } + return { ok: true }; +} diff --git a/apps/worker/test-dist-q/ai/cache.js b/apps/worker/test-dist-q/ai/cache.js new file mode 100644 index 00000000..850329c6 --- /dev/null +++ b/apps/worker/test-dist-q/ai/cache.js @@ -0,0 +1,33 @@ +const PREFIX = "ai-cache"; +export async function sha256Hex(input) { + const buf = new TextEncoder().encode(input); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} +export async function aiCacheGet(env, key) { + if (!env.AI_CACHE) + return null; + try { + const obj = await env.AI_CACHE.get(`${PREFIX}/${key}`); + if (!obj) + return null; + const text = await obj.text(); + return JSON.parse(text); + } + catch { + return null; + } +} +export async function aiCachePut(env, key, value) { + if (!env.AI_CACHE) + return; + try { + await env.AI_CACHE.put(`${PREFIX}/${key}`, JSON.stringify(value), { + httpMetadata: { contentType: "application/json" }, + customMetadata: { stored_at: new Date().toISOString() }, + }); + } + catch { + /* swallow — cache is best-effort */ + } +} diff --git a/apps/worker/test-dist-q/ai/extract.js b/apps/worker/test-dist-q/ai/extract.js new file mode 100644 index 00000000..fdde5b67 --- /dev/null +++ b/apps/worker/test-dist-q/ai/extract.js @@ -0,0 +1,335 @@ +// AI-powered extraction (Task #25 step 2). +// +// Strategy: deterministic strategies in `firmcrawl/personExtract.ts` run +// first (cheap). Misses get routed through Workers AI with a strict JSON +// schema response, chunked at ~6KB. Every call is cached by +// sha256(model+prompt+chunk) in the AI_CACHE R2 bucket (30-day TTL via +// bucket lifecycle policy). Verification pass drops <0.6 confidence. +// +// Hooked into pipeline.ts firm_team_crawl path opportunistically: if `AI` +// binding is present and deterministic strategies returned <3 people on a +// non-trivial page, we run the AI pass and union by nameKey. +import { aiCacheGet, aiCachePut, sha256Hex } from "./cache"; +import { assertBudget } from "./budget"; +import { limitAi } from "../scraper/rateLimit"; +import { trackAi } from "../analytics/events"; +// Task #2: hard timeout for Workers AI calls. The binding does not accept +// AbortSignal, so we race against a timer and surface a uniform +// "ai_timeout" error. +// +// POLICY: one canonical 30s ceiling for any single AI call (constant +// `AI_TIMEOUT_MS` below). This is well below the 90s default job +// budget, so even three serial AI calls fit inside a single job's +// wall-clock ceiling. Short-form purposes (embeddings, arbitration) +// use `AI_TIMEOUT_SHORT_MS` (20s) since they're trivially smaller. +// +// NB: this is a *caller-side* timeout — it bounds how long the worker +// will wait on the Workers AI binding, but it does NOT cancel the +// underlying model execution. The binding doesn't expose an +// AbortSignal as of this revision, so model inference may continue +// (and bill) for a short tail after we move on. Acceptable today +// because the queue-level budget + sweeper will reclaim the job; if +// the binding gains cancellation, swap the race for a real abort. +const AI_TIMEOUT_MS = 30_000; +const AI_TIMEOUT_SHORT_MS = 20_000; +async function runAiWithTimeout(p, ms, label) { + let timer = null; + try { + return await Promise.race([ + p, + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error(`ai_timeout:${label}:${ms}ms`)), ms); + }), + ]); + } + finally { + if (timer) + clearTimeout(timer); + } +} +const PERSON_SCHEMA = { + type: "object", + properties: { + people: { + type: "array", + items: { + type: "object", + properties: { + name: { type: "string" }, + role: { type: "string" }, + email: { type: "string" }, + linkedin: { type: "string" }, + twitter: { type: "string" }, + bio: { type: "string" }, + confidence: { type: "number" }, + }, + required: ["name", "confidence"], + }, + }, + }, + required: ["people"], +}; +const CHUNK_BYTES = 6000; +const MIN_CONFIDENCE = 0.6; +function chunk(text, size) { + const out = []; + for (let i = 0; i < text.length; i += size) + out.push(text.slice(i, i + size)); + return out; +} +function stripHtml(html) { + return html + .replace(/]*>[\s\S]*?<\/script>/gi, " ") + .replace(/]*>[\s\S]*?<\/style>/gi, " ") + .replace(/<[^>]+>/g, " ") + .replace(/\s+/g, " ") + .trim(); +} +export async function aiExtractPeople(env, html, jobId) { + if (!env.AI) + return []; + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return []; + if (!(await limitAi(env))) + return []; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const text = stripHtml(html); + const chunks = chunk(text, CHUNK_BYTES).slice(0, 4); // hard ceiling per page + const all = []; + for (const c of chunks) { + const cacheKey = await sha256Hex(`${model}:people:${c}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "extraction", model, cacheHit: true, jobId }); + all.push(...cached); + continue; + } + const t0 = Date.now(); + let people = []; + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "Extract investors/partners as JSON. Skip non-people. Return strict JSON." }, + { role: "user", content: `Extract people from this team-page text. ${c}` }, + ], + response_format: { type: "json_schema", json_schema: PERSON_SCHEMA }, + }), AI_TIMEOUT_MS, "extract_people")); + const parsed = parsePeopleResponse(res); + people = parsed.filter((p) => (p.confidence ?? 0) >= MIN_CONFIDENCE); + } + catch (e) { + console.warn("aiExtractPeople failed", e.message); + } + trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(c.length), jobId }); + await aiCachePut(env, cacheKey, people); + all.push(...people); + } + return dedupePeopleByName(all); +} +function parsePeopleResponse(res) { + const r = res; + if (Array.isArray(r?.people)) + return r.people.filter((p) => p && typeof p.name === "string"); + if (typeof r?.response === "string") { + try { + const j = JSON.parse(r.response); + if (Array.isArray(j?.people)) + return j.people.filter((p) => p && typeof p.name === "string"); + } + catch { /* fall through */ } + } + return []; +} +function dedupePeopleByName(arr) { + const map = new Map(); + for (const p of arr) { + const key = p.name.trim().toLowerCase(); + if (!key) + continue; + const cur = map.get(key); + if (!cur || (p.confidence ?? 0) > (cur.confidence ?? 0)) + map.set(key, p); + } + return [...map.values()]; +} +// Rough neurons estimate: tokens ≈ chars/4, llama-3.1-8b ~ 0.011 neurons/token. +function estimateNeurons(chars) { + const tokens = Math.ceil(chars / 4); + return Math.round(tokens * 0.011 * 1000) / 1000; +} +const TABLE_SCHEMA = { + type: "object", + properties: { + tables: { + type: "array", + items: { + type: "object", + properties: { + headers: { type: "array", items: { type: "string" } }, + rows: { type: "array", items: { type: "array", items: { type: "string" } } }, + }, + required: ["headers", "rows"], + }, + }, + }, + required: ["tables"], +}; +const TABLE_PAGE_CHAR_CAP = 8000; +const MAX_AI_PAGES = 12; +export async function aiExtractTablesFromPdfPages(env, pageTexts) { + if (!env.AI || !pageTexts.length) + return []; + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return []; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const out = []; + let lastHeaderKey = null; + const pages = pageTexts.slice(0, MAX_AI_PAGES); + for (let p = 0; p < pages.length; p++) { + const raw = pages[p].trim(); + if (raw.length < 40) + continue; + const text = raw.length > TABLE_PAGE_CHAR_CAP ? raw.slice(0, TABLE_PAGE_CHAR_CAP) : raw; + const cacheKey = await sha256Hex(`${model}:pdf-tables:${text}`); + let pageTables = await aiCacheGet(env, cacheKey); + if (pageTables) { + trackAi(env, { purpose: "extraction", model, cacheHit: true }); + } + else { + if (!(await limitAi(env))) + continue; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "You extract tabular data from a single PDF page. Return strict JSON {tables: [{headers, rows}]}. Skip page numbers, app chrome (File/Edit/View toolbars, sheet tab strips, Share buttons), and prose paragraphs. If the page has no table, return {tables: []}. Each row must have the same length as headers; pad with empty strings if needed." }, + { role: "user", content: `PDF page text:\n${text}` }, + ], + response_format: { type: "json_schema", json_schema: TABLE_SCHEMA }, + }), AI_TIMEOUT_MS, "extract_tables")); + pageTables = parseTablesResponse(res); + } + catch (e) { + console.warn("aiExtractTablesFromPdfPages failed", e.message); + pageTables = []; + } + trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); + await aiCachePut(env, cacheKey, pageTables); + } + for (const t of pageTables) { + if (!Array.isArray(t.headers) || t.headers.length < 2) + continue; + if (!Array.isArray(t.rows) || t.rows.length < 1) + continue; + const headers = t.headers.map((h) => String(h || "").trim()); + const headerKey = headers.join("|").toLowerCase(); + const rows = t.rows.map((r) => { + const obj = {}; + for (let c = 0; c < headers.length; c++) + obj[headers[c] || `col_${c}`] = String(r?.[c] ?? "").trim(); + return obj; + }).filter((r) => Object.values(r).some((v) => v.length > 0)); + if (!rows.length) + continue; + if (lastHeaderKey === headerKey && out.length) { + out[out.length - 1].rows.push(...rows); + } + else { + out.push({ headers, rows, pageNumber: p + 1, confidence: 0.5 }); + lastHeaderKey = headerKey; + } + } + } + return out; +} +function parseTablesResponse(res) { + const r = res; + if (Array.isArray(r?.tables)) + return r.tables; + if (typeof r?.response === "string") { + try { + const j = JSON.parse(r.response); + if (Array.isArray(j?.tables)) + return j.tables; + } + catch { /* fall through */ } + } + return []; +} +export async function aiEmbed(env, text) { + if (!env.AI) + return null; + const model = env.AI_EMBED_MODEL ?? "@cf/baai/bge-base-en-v1.5"; + const cacheKey = await sha256Hex(`${model}:embed:${text}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "embedding", model, cacheHit: true }); + return cached; + } + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return null; + if (!(await limitAi(env))) + return null; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { text: [text] }), AI_TIMEOUT_SHORT_MS, "embed")); + const vec = Array.isArray(res?.data?.[0]) ? res.data[0] : null; + if (!vec) + return null; + trackAi(env, { purpose: "embedding", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); + await aiCachePut(env, cacheKey, vec); + return vec; + } + catch (e) { + console.warn("aiEmbed failed", e.message); + return null; + } +} +export async function aiArbitrate(env, candidateA, candidateB) { + if (!env.AI) + return { match: "maybe", confidence: 0 }; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const cacheKey = await sha256Hex(`${model}:arb:${candidateA}|${candidateB}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "arbitration", model, cacheHit: true }); + return cached; + } + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return { match: "maybe", confidence: 0 }; + if (!(await limitAi(env))) + return { match: "maybe", confidence: 0 }; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "Decide if two profiles describe the same person. Reply JSON: {match: yes|no|maybe, confidence: 0..1}." }, + { role: "user", content: `A: ${candidateA}\nB: ${candidateB}` }, + ], + response_format: { type: "json_object" }, + }), AI_TIMEOUT_SHORT_MS, "arbitrate")); + const out = parseArbResponse(res); + trackAi(env, { purpose: "arbitration", model, ms: Date.now() - t0, neurons: estimateNeurons(candidateA.length + candidateB.length) }); + await aiCachePut(env, cacheKey, out); + return out; + } + catch (e) { + console.warn("aiArbitrate failed", e.message); + return { match: "maybe", confidence: 0 }; + } +} +function parseArbResponse(res) { + if (typeof res?.response === "string") { + try { + const j = JSON.parse(res.response); + const match = j.match === "yes" || j.match === "no" ? j.match : "maybe"; + return { match, confidence: Math.max(0, Math.min(1, Number(j.confidence ?? 0))) }; + } + catch { /* fall through */ } + } + return { match: "maybe", confidence: 0 }; +} diff --git a/apps/worker/test-dist-q/analytics/events.js b/apps/worker/test-dist-q/analytics/events.js new file mode 100644 index 00000000..c9fb991c --- /dev/null +++ b/apps/worker/test-dist-q/analytics/events.js @@ -0,0 +1,79 @@ +// Analytics Engine event helpers (Task #25 step 7). +// +// Cloudflare Analytics Engine accepts up to 20 blobs (strings), 20 doubles, +// and 1 list of indexes per data point. We standardize the schema across all +// AI/fetch events so the GraphQL Analytics API queries stay simple. +// +// Schema: +// indexes: [purpose] e.g. "extraction" | "embedding" | "arbitration" | … +// blobs: [purpose, model, host, cache_hit, job_id] +// doubles: [neurons, ms, bytes, cost_usd] +export function trackAi(env, args) { + const purpose = args.purpose; + const cache = args.cacheHit ? "1" : "0"; + try { + env.ANALYTICS?.writeDataPoint({ + indexes: [purpose], + blobs: [purpose, args.model, "", cache, args.jobId ?? ""], + doubles: [args.neurons ?? 0, args.ms ?? 0, 0, args.costUsd ?? 0], + }); + } + catch { + /* analytics is best-effort */ + } + // Also persist a daily roll-up to D1 so /api/analytics/ae/ai-cost works + // without the GraphQL Analytics API (which requires an account-level token). + if (!args.cacheHit) { + void rollupAiCost(env, purpose, args.model, args.neurons ?? 0, args.costUsd ?? 0); + } +} +// Vectorize ops are billed per query/upsert independent of AI neurons. We +// count them in the same ai_cost_daily roll-up under purpose='vectorize_' +// so /api/scrapers/health can surface the daily burn against +// VECTORIZE_DAILY_QUERIES_CAP. +export function trackVectorize(env, args) { + try { + env.ANALYTICS?.writeDataPoint({ + indexes: [`vectorize_${args.op}`], + blobs: [`vectorize_${args.op}`, args.index, "", "0", ""], + doubles: [0, 0, 0, 0], + }); + } + catch { /* best-effort */ } + void rollupVectorizeOp(env, args.op, args.index); +} +async function rollupVectorizeOp(env, op, index) { + try { + const day = new Date().toISOString().slice(0, 10); + await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) + VALUES (?, ?, ?, 0, 0, 1) + ON CONFLICT(day, purpose, model) DO UPDATE SET calls = calls + 1`).bind(day, `vectorize_${op}`, index).run(); + } + catch (e) { + console.warn("vectorize rollup failed", e.message); + } +} +export function trackFetch(env, args) { + try { + env.ANALYTICS?.writeDataPoint({ + indexes: ["fetch"], + blobs: ["fetch", String(args.tier), args.host, args.blockReason ?? "", String(args.status)], + doubles: [0, args.ms, args.bytes, 0], + }); + } + catch { /* best-effort */ } +} +async function rollupAiCost(env, purpose, model, neurons, costUsd) { + try { + const day = new Date().toISOString().slice(0, 10); + await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) + VALUES (?, ?, ?, ?, ?, 1) + ON CONFLICT(day, purpose, model) DO UPDATE SET + neurons = neurons + excluded.neurons, + cost_usd = cost_usd + excluded.cost_usd, + calls = calls + 1`).bind(day, purpose, model, neurons, costUsd).run(); + } + catch (e) { + console.warn("ai_cost_daily rollup failed", e.message); + } +} diff --git a/apps/worker/test-dist-q/db/error_log.js b/apps/worker/test-dist-q/db/error_log.js new file mode 100644 index 00000000..d4ae5fec --- /dev/null +++ b/apps/worker/test-dist-q/db/error_log.js @@ -0,0 +1,125 @@ +// Task #27: structured error logging into D1 + Analytics Engine mirror. +// +// Best-effort writes — we never let a logging failure mask the original +// error. Truncation rules keep us under D1's 1MB row cap. Every call also +// emits one Analytics Engine data point so spike detection / alerting can +// be done without scanning D1. +import { AppError, wrapUnknown } from "../errors"; +const MAX_MSG = 2000; +const MAX_STACK = 8000; +const MAX_CONTEXT_JSON = 16000; +function clip(s, max) { + if (!s) + return null; + return s.length > max ? s.slice(0, max) + "…[clipped]" : s; +} +function hostFromUrl(u) { + if (!u) + return null; + try { + return new URL(u).hostname.toLowerCase(); + } + catch { + return null; + } +} +export async function logError(env, input) { + const e = input.err instanceof AppError ? input.err : wrapUnknown(input.err, "internal_error"); + const host = input.host ?? hostFromUrl(input.url); + // Mirror to Analytics Engine first (cheap, doesn't depend on D1). + try { + if (env.ANALYTICS) { + env.ANALYTICS.writeDataPoint({ + indexes: [e.code], + blobs: [ + e.kind, + e.code, + input.step ?? "", + input.job_id ?? "", + input.request_id ?? "", + host ?? "", + input.workflow_run_id ?? "", + input.user_email ?? "", + ], + doubles: [e.status, e.retryable ? 1 : 0, input.retry_count ?? 0], + }); + } + } + catch { /* never throw from logger */ } + if (!env.DB) + return null; + let contextJson = null; + try { + contextJson = clip(JSON.stringify(e.context ?? {}), MAX_CONTEXT_JSON); + } + catch { + contextJson = null; + } + try { + const r = await env.DB.prepare(`INSERT INTO error_log + (request_id, job_id, step, code, kind, status, retryable, message, context_json, + cause_name, cause_message, cause_stack, url, method, + workflow_run_id, host, user_email, retry_count) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) + .bind(input.request_id ?? null, input.job_id ?? null, input.step ?? null, e.code, e.kind, e.status, e.retryable ? 1 : 0, clip(e.message, MAX_MSG), contextJson, e.cause?.name ?? null, clip(e.cause?.message, MAX_MSG), clip(e.cause?.stack, MAX_STACK), input.url ?? null, input.method ?? null, input.workflow_run_id ?? null, host, input.user_email ?? null, input.retry_count ?? 0) + .run(); + const id = r.meta?.last_row_id; + return typeof id === "number" ? id : null; + } + catch (logErr) { + // Never throw from the logger. + // Telemetry-of-telemetry: never throw, never log to console (CI gate). + void logErr; + return null; + } +} +export async function logStep(env, input) { + if (!env.DB) + return; + let metaJson = null; + try { + metaJson = input.meta ? JSON.stringify(input.meta) : null; + } + catch { + metaJson = null; + } + const finishedAt = input.status === "started" ? null : new Date().toISOString(); + try { + await env.DB.prepare(`INSERT INTO workflow_step_log + (job_id, step, step_name, status, finished_at, duration_ms, count_in, count_out, error_id, meta_json, workflow_run_id, attempt, error_code) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) + .bind(input.job_id, input.step, input.step, input.status, finishedAt, input.duration_ms ?? null, input.count_in ?? null, input.count_out ?? null, input.error_id ?? null, metaJson, input.workflow_run_id ?? null, input.attempt ?? 1, input.error_code ?? null) + .run(); + } + catch (e) { + void e; + } +} +/** Convenience: time a function and log start+finish to workflow_step_log. */ +export async function timedStep(env, job_id, step, fn, opts = {}) { + const t0 = Date.now(); + const { workflow_run_id, attempt } = opts; + await logStep(env, { job_id, step, status: "started", count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt }); + try { + const out = await fn(); + const count_out = Array.isArray(out) ? out.length : undefined; + await logStep(env, { job_id, step, status: "ok", duration_ms: Date.now() - t0, count_in: opts.count_in, count_out, meta: opts.meta, workflow_run_id, attempt }); + return out; + } + catch (e) { + const { classify, isBenignSkip } = await import("../errors.js"); + // Task #72: robots.txt / ToS blocks are benign policy skips, not errors. + // Don't write an error_log row (it would surface as a red 422 in the + // operator console) — record the step as `skipped` instead and rethrow so + // the caller routes the job to the `skipped` terminal status. + const skip = isBenignSkip(e); + if (skip) { + await logStep(env, { job_id, step, status: "skipped", duration_ms: Date.now() - t0, count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt, error_code: skip.skip_code }); + throw e; + } + const error_id = await logError(env, { err: e, job_id, step }); + const cls = classify(e); + await logStep(env, { job_id, step, status: "error", duration_ms: Date.now() - t0, count_in: opts.count_in, error_id, meta: opts.meta, workflow_run_id, attempt, error_code: cls?.code }); + throw e; + } +} diff --git a/apps/worker/test-dist-q/entities/channels.js b/apps/worker/test-dist-q/entities/channels.js new file mode 100644 index 00000000..41a13c4c --- /dev/null +++ b/apps/worker/test-dist-q/entities/channels.js @@ -0,0 +1,61 @@ +// Channel upsert keyed by (entity_id, kind, canonical). Canonical form +// is computed by ./normalize so trivially-different inputs collapse. +import { canonicalEmail, canonicalPhone, canonicalLinkedin, canonicalTwitter, canonicalGithub, canonicalUrl, } from "./normalize"; +export function canonicalizeFor(kind, raw) { + switch (kind) { + case "email": return canonicalEmail(raw); + case "phone": return canonicalPhone(raw); + case "linkedin": return canonicalLinkedin(raw); + case "twitter": return canonicalTwitter(raw); + case "github": return canonicalGithub(raw); + case "website": + case "other": + return canonicalUrl(raw); + } +} +export async function upsertChannel(env, input) { + if (!input.entity_id || !input.canonical) + return null; + // Re-canonicalize defensively in case caller passed a raw value. + const canonical = canonicalizeFor(input.kind, input.canonical) ?? input.canonical; + const now = new Date().toISOString(); + const existing = await env.DB.prepare(`SELECT id FROM channels WHERE entity_id = ? AND kind = ? AND canonical = ?`).bind(input.entity_id, input.kind, canonical).first(); + if (existing) { + const sets = ["last_seen_at = ?"]; + const binds = [now]; + if (input.display) { + sets.push("display = COALESCE(display, ?)"); + binds.push(input.display); + } + if (input.is_primary) { + sets.push("is_primary = 1"); + } + if (input.is_verified) { + sets.push("is_verified = 1"); + } + if (input.is_dnc) { + sets.push("is_dnc = 1"); + } + if (input.source) { + sets.push("source = COALESCE(source, ?)"); + binds.push(input.source); + } + binds.push(existing.id); + await env.DB.prepare(`UPDATE channels SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); + return existing.id; + } + const id = crypto.randomUUID(); + await env.DB.prepare(`INSERT INTO channels (id, entity_id, kind, canonical, display, is_primary, is_verified, is_dnc, source, confidence, first_seen_at, last_seen_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`).bind(id, input.entity_id, input.kind, canonical, input.display ?? null, input.is_primary ? 1 : 0, input.is_verified ? 1 : 0, input.is_dnc ? 1 : 0, input.source ?? null, input.confidence ?? 1, now, now).run(); + return id; +} +export async function findEntityByChannel(env, kind, raw) { + const canonical = canonicalizeFor(kind, raw); + if (!canonical) + return null; + const r = await env.DB.prepare(`SELECT c.entity_id FROM channels c + JOIN u_entities e ON e.id = c.entity_id + WHERE c.kind = ? AND c.canonical = ? AND e.status NOT IN ('merged','soft_deleted') + ORDER BY c.is_primary DESC, c.is_verified DESC, c.last_seen_at DESC LIMIT 1`).bind(kind, canonical).first(); + return r?.entity_id ?? null; +} diff --git a/apps/worker/test-dist-q/entities/dualwrite.js b/apps/worker/test-dist-q/entities/dualwrite.js new file mode 100644 index 00000000..505a51bc --- /dev/null +++ b/apps/worker/test-dist-q/entities/dualwrite.js @@ -0,0 +1,375 @@ +// Dual-write hooks. Every legacy writer calls one of these to mirror the +// row into the unified entity model. All hooks are best-effort: they +// log + swallow errors so a transient unified-model failure never blocks +// the legacy path. +import { createEntity, addRole, getLegacyEntityId, setLegacyEntityId } from "./roles"; +import { insertFactsBatch } from "./facts"; +import { upsertChannel, findEntityByChannel } from "./channels"; +import { addTag, addTagsFromJsonArray } from "./tags"; +import { canonicalEmail, canonicalLinkedin, canonicalDomain } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +async function resolveOrCreate(env, table, legacyId, kind, init, channelLookups) { + const existing = await getLegacyEntityId(env, table, legacyId); + if (existing) + return existing; + // Cross-link by strongest available identifier before creating new: + // (a) deterministic domain match against u_entities.primary_domain + // (orgs only — collapses firm/account/company duplicates sharing a + // domain); (b) primary_email_key/primary_linkedin_key direct hits; + // (c) channel-table reverse lookups for any other handle. + if (kind === "org" && init.primary_domain) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_domain = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_domain).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + if (kind === "person" && init.primary_email_key) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_email_key = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_email_key).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + if (init.primary_linkedin_key) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_linkedin_key = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_linkedin_key).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + for (const ch of channelLookups) { + if (!ch.raw) + continue; + const hit = await findEntityByChannel(env, ch.kind, ch.raw); + if (hit) { + await setLegacyEntityId(env, table, legacyId, hit); + return hit; + } + } + try { + const created = await createEntity(env, { kind, ...init }); + if (!created) + return null; // Task #9: rejected by garbage detector + await setLegacyEntityId(env, table, legacyId, created.id); + return created.id; + } + catch (e) { + console.warn("dualwrite createEntity failed", table, legacyId, e.message); + return null; + } +} +function chan(env, entityId, kind, raw, source, primary = false) { + if (!raw) + return Promise.resolve(null); + return upsertChannel(env, { entity_id: entityId, kind, canonical: String(raw), source, is_primary: primary }); +} +export async function syncFirmToEntity(env, f, source = "firms_upsert", sourceKind = "scrape") { + try { + const domain = canonicalDomain(f.domain ?? f.website); + const linkedin = canonicalLinkedin(f.linkedin_url); + const entityId = await resolveOrCreate(env, "firms", f.id, "org", { + display_name: f.name, + primary_domain: domain, + primary_url: f.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: f.linkedin_url }, + { kind: "website", raw: f.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "firm", { is_primary: true, source }); + if (f.kind && /accelerator/i.test(f.kind)) + await addRole(env, entityId, "accelerator", { source }); + // Task #2: cascading role inference on the unified write path so + // investor_firm / customer / prospect get assigned without each + // call-site having to know. + try { + const { inferAndAssignRoles } = await import("../services/roleInference.js"); + await inferAndAssignRoles(env, entityId, { + kind: "org", + sourceKind, + sourceUrl: f.website ?? null, + sourceDomain: domain ?? null, + org: f.name ?? null, + category: f.kind ?? null, + importLabel: source, + }); + } + catch (e) { + console.warn("inferAndAssignRoles(firm) failed", entityId, e.message); + } + const patches = [ + { predicate: "name", value_text: f.name }, + { predicate: "legal_name", value_text: f.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: f.website ?? null }, + { predicate: "country_iso2", value_text: f.hq_country_iso2 ?? null }, + { predicate: "region", value_text: f.hq_region ?? null }, + { predicate: "city", value_text: f.hq_city ?? null }, + { predicate: "thesis", value_text: f.thesis ?? null }, + { predicate: "kind", value_text: f.kind ?? null }, + { predicate: "check_size_min_usd", value_number: numOrNull(f.check_size_min_usd) }, + { predicate: "check_size_max_usd", value_number: numOrNull(f.check_size_max_usd) }, + { predicate: "check_size_typical_usd", value_number: numOrNull(f.check_size_typical_usd) }, + ]; + await insertFactsBatch(env, entityId, patches, source, sourceKind); + await Promise.all([ + chan(env, entityId, "website", f.website, source, true), + chan(env, entityId, "linkedin", f.linkedin_url, source, true), + chan(env, entityId, "other", f.crunchbase_url, source), + chan(env, entityId, "twitter", f.twitter_handle, source), + chan(env, entityId, "email", f.contact_email, source), + ]); + await Promise.all([ + addTagsFromJsonArray(env, entityId, "sector", f.sectors_json, source), + addTagsFromJsonArray(env, entityId, "stage", f.stages_json, source), + addTagsFromJsonArray(env, entityId, "geo", f.geo_focus_json, source), + ]); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncFirmToEntity failed", f.id, e.message); + return null; + } +} +export async function syncLeadToEntity(env, l, source = "leads_repo", sourceKind = "scrape") { + try { + const email = canonicalEmail(l.email); + const linkedin = canonicalLinkedin(l.linkedin_url); + const entityId = await resolveOrCreate(env, "leads", l.id, "person", { + display_name: l.name ?? null, + primary_email_key: email, + primary_linkedin_key: linkedin, + primary_url: l.personal_url ?? null, + }, [ + { kind: "email", raw: l.email }, + { kind: "linkedin", raw: l.linkedin_url }, + ]); + if (!entityId) + return null; + // Role inference: investor_kind set → investor; otherwise generic person. + if (l.investor_kind) + await addRole(env, entityId, "investor", { is_primary: true, source }); + if (l.category && /founder|ceo|cto/i.test(l.category)) + await addRole(env, entityId, "founder", { source }); + // Task #2: cascading role inference on the unified write path. Runs + // for every person sync so the Investors page (which reads from + // entity_roles) picks up freshly-ingested people. Looks up the + // person's company entity for partner-title inheritance. + try { + const { inferAndAssignRoles } = await import("../services/roleInference.js"); + await inferAndAssignRoles(env, entityId, { + kind: "person", + sourceKind, + sourceUrl: l.source_url ?? l.personal_url ?? null, + sourceDomain: l.source_domain ?? null, + title: l.title ?? null, + org: l.org ?? null, + category: l.category ?? null, + importLabel: source, + }); + } + catch (e) { + console.warn("inferAndAssignRoles(lead) failed", entityId, e.message); + } + const patches = [ + { predicate: "name", value_text: l.name ?? null }, + { predicate: "title", value_text: l.title ?? null }, + { predicate: "primary_employer", value_text: l.org ?? null }, + { predicate: "category", value_text: l.category ?? null }, + { predicate: "country_iso2", value_text: l.country_iso2 ?? null }, + { predicate: "region", value_text: l.region ?? null }, + { predicate: "city", value_text: l.city ?? null }, + { predicate: "bio", value_text: l.bio ?? null }, + { predicate: "thesis", value_text: l.thesis ?? null }, + { predicate: "investor_kind", value_text: l.investor_kind ?? null }, + { predicate: "check_size_min_usd", value_number: numOrNull(l.check_size_min_usd) }, + { predicate: "check_size_max_usd", value_number: numOrNull(l.check_size_max_usd) }, + { predicate: "check_size_typical_usd", value_number: numOrNull(l.check_size_typical_usd) }, + ]; + await insertFactsBatch(env, entityId, patches, source, sourceKind); + await Promise.all([ + chan(env, entityId, "email", l.email, source, true), + chan(env, entityId, "phone", l.phone, source), + chan(env, entityId, "linkedin", l.linkedin_url, source, true), + chan(env, entityId, "twitter", l.twitter_url, source), + chan(env, entityId, "github", l.github_url, source), + chan(env, entityId, "website", l.personal_url, source), + ]); + await Promise.all([ + addTagsFromJsonArray(env, entityId, "sector", l.sector_focus_json, source), + addTagsFromJsonArray(env, entityId, "stage", l.stage_focus_json, source), + addTagsFromJsonArray(env, entityId, "geo", l.geo_focus_json, source), + addTagsFromJsonArray(env, entityId, "tag", l.tags_json, source), + ]); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncLeadToEntity failed", l.id, e.message); + return null; + } +} +export async function syncCompanyToEntity(env, c, source = "companies") { + try { + const domain = canonicalDomain(c.domain ?? c.website); + const linkedin = canonicalLinkedin(c.linkedin_url); + const entityId = await resolveOrCreate(env, "companies", c.id, "org", { + display_name: c.name, + primary_domain: domain, + primary_url: c.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: c.linkedin_url }, + { kind: "website", raw: c.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "company", { is_primary: true, source }); + const patches = [ + { predicate: "name", value_text: c.name }, + { predicate: "legal_name", value_text: c.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: c.website ?? null }, + { predicate: "country_iso2", value_text: c.hq_country_iso2 ?? null }, + { predicate: "region", value_text: c.hq_region ?? null }, + { predicate: "city", value_text: c.hq_city ?? null }, + { predicate: "stage", value_text: c.stage ?? null }, + { predicate: "unicorn_count", value_number: c.unicorn ? 1 : 0 }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "website", c.website, source, true), + chan(env, entityId, "linkedin", c.linkedin_url, source), + chan(env, entityId, "twitter", c.twitter_handle, source), + chan(env, entityId, "github", c.github_org, source), + chan(env, entityId, "other", c.crunchbase_url, source), + ]); + await addTagsFromJsonArray(env, entityId, "sector", c.industries_json, source); + if (c.stage) + await addTag(env, { entity_id: entityId, taxonomy: "stage", slug: c.stage, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncCompanyToEntity failed", c.id, e.message); + return null; + } +} +export async function syncAccountToEntity(env, a, source = "accounts") { + try { + const domain = canonicalDomain(a.domain ?? a.website); + const linkedin = canonicalLinkedin(a.linkedin_url); + const entityId = await resolveOrCreate(env, "accounts", a.id, "org", { + display_name: a.name, + primary_domain: domain, + primary_url: a.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: a.linkedin_url }, + { kind: "website", raw: a.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "account", { is_primary: true, source }); + const patches = [ + { predicate: "name", value_text: a.name }, + { predicate: "legal_name", value_text: a.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: a.website ?? null }, + { predicate: "country_iso2", value_text: a.hq_country_iso2 ?? null }, + { predicate: "region", value_text: a.hq_region ?? null }, + { predicate: "city", value_text: a.hq_city ?? null }, + { predicate: "industry", value_text: a.industry ?? null }, + { predicate: "funding_stage", value_text: a.funding_stage ?? null }, + { predicate: "fit_max_score", value_number: numOrNull(a.fit_score) }, + { predicate: "intent_score", value_number: numOrNull(a.intent_score) }, + // `accounts.employees` was the one sizing column dualwrite dropped, so + // no account entity carried a headcount fact and persona matching + // scored every one of them "company size unknown". `employees` is the + // predicate the registry declares (entities/profile-predicates.ts) and + // the one secEdgar/persist.ts already writes, so this converges on the + // name that exists rather than adding a fourth spelling. + { predicate: "employees", value_number: numOrNull(a.employees) }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "website", a.website, source, true), + chan(env, entityId, "linkedin", a.linkedin_url, source), + chan(env, entityId, "twitter", a.twitter_handle, source), + chan(env, entityId, "github", a.github_org, source), + chan(env, entityId, "other", a.crunchbase_url, source), + ]); + await addTagsFromJsonArray(env, entityId, "sector", a.industries_json, source); + if (a.industry) + await addTag(env, { entity_id: entityId, taxonomy: "sector", slug: a.industry, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncAccountToEntity failed", a.id, e.message); + return null; + } +} +export async function syncBuyerToEntity(env, b, source = "buyers") { + try { + const email = canonicalEmail(b.email); + const linkedin = canonicalLinkedin(b.linkedin_url); + const entityId = await resolveOrCreate(env, "buyers", b.id, "person", { + display_name: b.name ?? null, + primary_email_key: email, + primary_linkedin_key: linkedin, + }, [ + { kind: "email", raw: b.email }, + { kind: "linkedin", raw: b.linkedin_url }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "buyer", { is_primary: true, source }); + if (b.is_decision_maker) + await addRole(env, entityId, "executive", { source }); + // Link buyer → account via 'works_at' edge. + const accountEntityId = await getLegacyEntityId(env, "accounts", b.account_id); + if (accountEntityId) { + await env.DB.prepare(`INSERT INTO rel_edges (id, src_entity_id, dst_entity_id, kind, source) + VALUES (?, ?, ?, 'works_at', ?) + ON CONFLICT(src_entity_id, dst_entity_id, kind, IFNULL(valid_from,'')) DO NOTHING`).bind(crypto.randomUUID(), entityId, accountEntityId, source).run().catch(() => undefined); + // Also write as a fact for summary.primary_employer_entity_id pickup. + await insertFactsBatch(env, entityId, [{ predicate: "employer", value_entity_id: accountEntityId }], source, "inferred"); + } + const patches = [ + { predicate: "name", value_text: b.name ?? null }, + { predicate: "title", value_text: b.title ?? null }, + { predicate: "seniority", value_text: b.seniority ?? null }, + { predicate: "department", value_text: b.department ?? null }, + { predicate: "role_slug", value_text: b.role_slug ?? null }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "email", b.email, source, true), + chan(env, entityId, "linkedin", b.linkedin_url, source, true), + chan(env, entityId, "twitter", b.twitter_url, source), + chan(env, entityId, "phone", b.phone, source), + ]); + if (b.role_slug) + await addTag(env, { entity_id: entityId, taxonomy: "role", slug: b.role_slug, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncBuyerToEntity failed", b.id, e.message); + return null; + } +} +function numOrNull(v) { + return typeof v === "number" && Number.isFinite(v) ? v : null; +} diff --git a/apps/worker/test-dist-q/entities/facts.js b/apps/worker/test-dist-q/entities/facts.js new file mode 100644 index 00000000..a40ff5df --- /dev/null +++ b/apps/worker/test-dist-q/entities/facts.js @@ -0,0 +1,282 @@ +// Fact insertion with content-addressed dedup. The DB trigger +// `trg_facts_supersede` flips the prior is_current=1 fact for the same +// (entity, predicate, source) to 0 after insert. +import { sha256 } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +// Task #8: triggers debounced persona ↔ entity re-match when a fact +// that materially affects scoring is written. No-op otherwise. +import { triggerEntityMatchRefresh, isRelevantPredicate } from "../services/personaMatchTrigger"; +// Task #51: route previously-swallowed fact-write-path failures through the +// structured error logger so they land in `error_log` instead of vanishing. +import { logError } from "../db/error_log"; +export async function insertFact(env, f) { + if (!f.entity_id || !f.predicate) + return null; + const valueKey = JSON.stringify({ + t: f.value_text ?? null, + n: f.value_number ?? null, + j: f.value_json ?? null, + e: f.value_entity_id ?? null, + }); + const hash = await sha256(`${f.entity_id}|${f.predicate}|${valueKey}|${f.source ?? ""}`); + const id = crypto.randomUUID(); + const now = f.observed_at ?? new Date().toISOString(); + // Task #3 (Editable Profiles): lock check. If a locked override exists + // for this (entity, predicate), the new fact row is still inserted (so + // the diff strip can show the AI/scrape attempt) but stamped with + // superseded_by_override=1 so it never wins the read race. The override + // layer overlays at read time via getEffectiveFacts. + // Wrapped in try/catch (not just `.catch`) so a missing `field_overrides` + // table — fresh install ahead of migration 376, or a minimal test DB whose + // prepare() throws synchronously — is logged and degrades to "no lock" + // instead of failing the fact write. + const lock = await (async () => { + try { + return await env.DB.prepare(`SELECT 1 FROM field_overrides + WHERE entity_id = ? AND predicate = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + LIMIT 1`).bind(f.entity_id, f.predicate).first(); + } + catch (e) { + await logError(env, { err: e, step: "facts.insertFact.override_lock_check" }); + return null; + } + })(); + const supersededByOverride = lock ? 1 : 0; + try { + await env.DB.prepare(`INSERT INTO facts ( + id, entity_id, predicate, value_text, value_number, value_json, + value_entity_id, source_kind, source, evidence_url, confidence, + observed_at, valid_from, valid_to, is_current, hash, superseded_by_override + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 1, ?, ?)`).bind(id, f.entity_id, f.predicate, f.value_text ?? null, f.value_number ?? null, f.value_json != null ? JSON.stringify(f.value_json) : null, f.value_entity_id ?? null, f.source_kind, f.source ?? null, f.evidence_url ?? null, f.confidence ?? 1, now, f.valid_from ?? null, f.valid_to ?? null, hash, supersededByOverride).run(); + // Task #3 race fix: the SELECT lock-check above and this INSERT are + // not atomic. If an override landed between them, our row would have + // superseded_by_override=0 even though an override now dominates. The + // override-create handler ALSO runs `UPDATE facts SET + // superseded_by_override = 1` to catch facts inserted before the + // override; this post-insert re-check covers the reverse direction, + // so both writers converge on the same end state regardless of which + // raced first. + if (!supersededByOverride) { + await env.DB.prepare(`UPDATE facts SET superseded_by_override = 1 + WHERE id = ? + AND EXISTS ( + SELECT 1 FROM field_overrides + WHERE entity_id = ? AND predicate = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + )`).bind(id, f.entity_id, f.predicate).run().catch((e) => logError(env, { err: e, step: "facts.insertFact.override_recheck" })); + } + // Centralized rebuild guarantee: every successful fact insert + // enqueues a summary rebuild for the owning entity. This keeps the + // "fact INSERT → rebuild within ~5s" SLO honest regardless of which + // caller wrote the fact (dual-write, merge, manual admin, etc.). + await enqueueSummaryRebuild(env, f.entity_id); + if (isRelevantPredicate(f.predicate)) { + // Fire-and-forget; debounced via KV inside the trigger. + void triggerEntityMatchRefresh(env, f.entity_id).catch((e) => { + console.warn("triggerEntityMatchRefresh from insertFact failed", e.message); + }); + } + // Task #4 (Relationship Inference Worker): debounced enqueue into + // relationship_infer_queue (migration 377). KV-debounced 60s; the + // consolidated nightly slot drains the queue with the per-entity + // orchestrator pass. Never inline — entity/fact writes stay fast. + // No `relationship_infer` JobKind exists, so we fall back to the + // nightly tick per the spec's explicit instruction. + try { + const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); + void enqueueRelInfer(env, f.entity_id, `fact:${f.predicate}`).catch((e) => logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer" })); + } + catch (e) { + void logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer_import" }); + } + return id; + } + catch (e) { + // UNIQUE(hash) collision = exact-replay observation. Task #1 + // requires re-imports of the same Folk row to refresh `observed_at` + // so freshness queries reflect when we last *saw* the fact, even + // when nothing about the value changed. We update the existing row + // (matched by hash) instead of writing a new one. + const msg = e.message || ""; + if (/UNIQUE/i.test(msg)) { + try { + await env.DB.prepare("UPDATE facts SET observed_at = ? WHERE hash = ?").bind(now, hash).run(); + } + catch (uErr) { + console.warn("insertFact observed_at refresh failed", uErr.message); + } + return null; + } + throw e; + } +} +function parseJsonSafe(s) { + if (s == null) + return null; + try { + return JSON.parse(s); + } + catch { + return s; + } +} +export async function loadCurrentOverrides(env, entityId) { + const r = await env.DB.prepare(`SELECT id, predicate, value_text, value_numeric, value_json, overridden_at + FROM field_overrides + WHERE entity_id = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + ORDER BY overridden_at DESC`).bind(entityId).all().catch(() => ({ results: [] })); + const map = new Map(); + for (const o of r.results ?? []) { + if (!map.has(o.predicate)) + map.set(o.predicate, o); + } + return map; +} +export async function getEffectiveFacts(env, entityId, opts) { + const factWhere = opts?.includeNonCurrent ? "" : " AND is_current = 1"; + const limit = opts?.limit ?? 500; + const [factsRes, overrides] = await Promise.all([ + env.DB.prepare(`SELECT id, predicate, value_text, value_number, value_json, value_entity_id, + source_kind, source, confidence, verified_score, observed_at, + is_current, superseded_by_override + FROM facts + WHERE entity_id = ?${factWhere} + ORDER BY observed_at DESC LIMIT ?`).bind(entityId, limit).all(), + loadCurrentOverrides(env, entityId), + ]); + const out = []; + const overridePredsSeen = new Set(); + for (const f of factsRes.results ?? []) { + const ov = overrides.get(f.predicate); + if (ov) { + if (!overridePredsSeen.has(f.predicate)) { + overridePredsSeen.add(f.predicate); + out.push({ + id: `override:${ov.id}`, + predicate: f.predicate, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: parseJsonSafe(ov.value_json), + value_entity_id: null, + source_kind: "manual", + source: "field_override", + confidence: 1, + verified_score: null, + observed_at: ov.overridden_at, + is_current: 1, + superseded_by_override: 0, + is_override: true, + override_id: ov.id, + overridden_attempt: false, + }); + } + // Mark the underlying fact as an overridden attempt for the diff + // strip. Never returned as canonical. + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 1, + is_override: false, + override_id: null, + overridden_attempt: true, + }); + } + else if (f.superseded_by_override === 1) { + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 1, + is_override: false, + override_id: null, + overridden_attempt: true, + }); + } + else { + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 0, + is_override: false, + override_id: null, + overridden_attempt: false, + }); + } + } + // Overrides for predicates with no underlying fact row at all. + for (const [pred, ov] of overrides.entries()) { + if (overridePredsSeen.has(pred)) + continue; + out.push({ + id: `override:${ov.id}`, + predicate: pred, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: parseJsonSafe(ov.value_json), + value_entity_id: null, + source_kind: "manual", + source: "field_override", + confidence: 1, + verified_score: null, + observed_at: ov.overridden_at, + is_current: 1, + superseded_by_override: 0, + is_override: true, + override_id: ov.id, + overridden_attempt: false, + }); + } + return out; +} +export async function insertFactsBatch(env, entityId, patches, source, sourceKind = "scrape", evidenceUrl = null) { + let n = 0; + for (const p of patches) { + if (p.value_text == null && p.value_number == null && p.value_json == null && p.value_entity_id == null) + continue; + const id = await insertFact(env, { + entity_id: entityId, + predicate: p.predicate, + value_text: p.value_text ?? null, + value_number: p.value_number ?? null, + value_json: p.value_json, + value_entity_id: p.value_entity_id ?? null, + source_kind: sourceKind, + source, + evidence_url: evidenceUrl, + }); + if (id) + n += 1; + } + return n; +} diff --git a/apps/worker/test-dist-q/entities/garbage.js b/apps/worker/test-dist-q/entities/garbage.js new file mode 100644 index 00000000..ffa95393 --- /dev/null +++ b/apps/worker/test-dist-q/entities/garbage.js @@ -0,0 +1,652 @@ +// Task #9: Garbage Entity Detector & Cleanup. +// +// Pure detector (`isGarbage`) flags HTML page titles / nav fragments / +// UI strings polluting `u_entities`. Used by: +// * The pre-insert guard in `createEntity` (rejects before write). +// * The cron sweep (`runCleanupSweep`) that soft-deletes recently- +// created garbage and, on `mode='all'`, performs the one-off pass. +// * The /ops/garbage-review/ console (admin restore / purge). +// +// HONEST DEGRADATION (Task #14 pattern): the optional Workers AI second +// opinion (`aiSecondOpinion`) returns `uncertain` when the `env.AI` +// binding is absent, on any HTTP/network error, or when the JSON +// response is malformed. `evaluateEntity` then DOES NOT flag the +// entity — never silently garbage. +const NAME_MAX_LEN = 80; +// Curated UI / nav strings observed in production on the Investors page. +// Lowercased for comparison. +const KNOWN_UI_STRINGS = new Set([ + "contact us", "contact", "search icon", "search", "home", "about", + "about us", "menu", "login", "log in", "sign in", "sign up", + "sign-up", "register", "our team", "team", "limited partners", + "portfolio", "our portfolio", "careers", "jobs", "privacy", + "privacy policy", "terms", "terms of service", "cookies", + "cookie policy", "blog", "news", "press", "press releases", + "get in touch", "subscribe", "newsletter", "footer", "header", + "navigation", "nav", "skip to content", "back to top", "read more", + "learn more", "view all", "see all", "all rights reserved", + "follow us", "share", "tweet", "facebook", "twitter", "linkedin", + "instagram", "youtube", "the team", "our story", +]); +// Heuristic leaders that strongly indicate a press/blog post title +// got captured as an "entity". +const LEADER_RE = /^(introducing|announcing|welcome to|how|why|what|when|where|the future of|inside)\s+/i; +// Page-title with `|`-separated domain/brand fragment. Examples: +// "Our Team | Sequoia Capital", "Contact Tenity | Get in Touch", +// "Home | Sequoia Capital". +const PIPE_TITLE_RE = /\s\|\s\S/; +// Pure emoji / icon names (no alphanumerics at all). +const NO_ALNUM_RE = /^[^\p{L}\p{N}]+$/u; +// Listicle / directory page titles captured as entity names. +// +// This is the gap that let ~128 non-firms into the firms table: a crawler +// ingested aggregator pages ("VC Firms By Stage" on failory.com) and made +// one entity per outbound link, taking the page title as the name. The +// existing rules could not catch it — such a title has no pipe fragment, no +// press leader, is well under 80 characters and is not a known nav string. +// +// Deliberately narrow, because a single matched reason marks an entity +// garbage. Each pattern is a phrase a real firm name essentially never +// contains: "Top Tier Capital Partners" and "Stage Fund" are real firms, so +// a bare leading "top" or the word "stage" alone must NOT match. +const LISTICLE_RES = [ + // "VC Firms By Stage", "Investors by sector", "Funds per geography" + /\b(?:by|per)\s+(?:stage|sector|industry|geograph|countr|region|check\s*size|vertical)/i, + // "Top 50 VC Firms", "Best 10 Seed Funds" — the number is what makes this + // safe; a leading "Top"/"Best" alone is a legitimate name fragment. + /^(?:the\s+)?(?:top|best|leading)\s+\d+\b/i, + // "List of European VCs", "Directory of angel investors" + /\b(?:list|directory|database|roundup|ranking)\s+of\s+/i, + // "The Ultimate Guide to Seed Funds", "Complete List of ..." + /^(?:the\s+)?(?:complete|ultimate|definitive)\s+(?:list|guide|directory|database)\b/i, +]; +// Minimum length before a name that is identical to its own domain slug is +// treated as URL-derived rather than a genuine single-word brand. +// "Stripe" / "Coatue" / "Atomico" are real names that equal their domain; +// "Firstmarkcap" (firstmarkcap.com) and "Forerunnerventures" are slugs that +// were title-cased because no real name was ever extracted. +const SLUG_NAME_MIN_LEN = 12; +/** + * True when the display name is just the registrable domain label with the + * first letter capitalised — i.e. the crawler never found a name and fell + * back to the URL. Length-gated so short single-word brands are untouched. + */ +function looksDomainDerived(name, domain) { + if (!domain) + return false; + const label = domain.toLowerCase().replace(/^www\./, "").split(".")[0] ?? ""; + if (label.length < SLUG_NAME_MIN_LEN) + return false; + const n = name.trim().toLowerCase(); + // A genuine name carries separators the slug cannot ("First Mark Capital"). + if (/[\s.\-_]/.test(n)) + return false; + return n === label; +} +// --------------------------------------------------------------------------- +// Task #6: person-name disambiguation. Classifies a name that was recorded +// as a `person` into one of: a real person, an organization scraped as a +// person (firm / fund / accelerator / company), generic page junk, or +// uncertain. Used by: +// * the pre-insert reclassify-on-write guard in `createEntity`, +// * the cron / one-off sweep (`runCleanupSweep`), +// * the scraper extraction boundary (`extractPeopleFromPage`). +// PURE — name-only, no IO — so it's safe on the hot write path. +// --------------------------------------------------------------------------- +// Legal-entity suffixes — an extremely strong organization signal anywhere +// in the name. Normalized (punctuation stripped) before comparison. +const ORG_LEGAL_SUFFIX = new Set([ + "llc", "inc", "ltd", "limited", "lp", "llp", "plc", "gmbh", "ag", + "sarl", "bv", "pty", "oy", "ab", "srl", "spa", +]); +// Descriptor words that, as the LAST token, denote an organization +// ("Intel Capital", "Mendoza Ventures", "Hillman Accelerator Foundation"). +const ORG_SUFFIX_LAST = new Set([ + "capital", "ventures", "venture", "partners", "partner", "holdings", + "group", "fund", "funds", "foundation", "labs", "lab", "hub", + "collective", "management", "advisors", "associates", "accelerator", + "incubator", "equity", "securities", "technologies", "studios", "studio", + "network", "institute", "academy", "council", "alliance", "syndicate", + "consortium", "enterprises", "industries", "international", "global", + "company", "corp", "corporation", "university", "college", "systems", + "solutions", +]); +// Generic, non-distinctive words. A name made up ENTIRELY of these is junk +// ("Deep Tech", "Our Mission"); they're also excluded when looking for a +// distinctive proper-noun token in an org name. +const GENERIC_WORDS = new Set([ + "the", "our", "your", "my", "a", "an", "all", "more", "new", "updated", + "featured", "latest", "recent", "top", "best", "of", "and", "or", "for", + "with", "to", "in", "on", "at", "by", "from", "about", "welcome", "hello", + "home", "homepage", "page", "web", "website", "webpage", "mission", + "vision", "values", "story", "team", "careers", "jobs", "blog", "news", + "press", "media", "map", "menu", "footer", "header", "sidebar", "gallery", + "resources", "events", "podcast", "newsletter", "insights", "research", + "report", "reports", "overview", "summary", "services", "solutions", + "products", "pricing", "features", "get", "started", "learn", "read", + "view", "see", "guide", "guides", "faq", "faqs", "help", "support", + "contact", "deep", "tech", "technology", "startup", "startups", + "mentorship", "money", "data", "signal", "community", "ecosystem", + "platform", "world", "global", "international", "region", "regions", + "area", "areas", "north", "south", "east", "west", "central", "america", + "americas", "europe", "asia", "africa", "oceania", "antarctica", "middle", + "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", + "thumbnail", "placeholder", "avatar", "headshot", "slideshow", "carousel", + "machine", "wayback", "future", "work", "working", "people", "portfolio", + "companies", "investors", "founders", "funding", "rounds", "deals", +]); +// Words whose presence alone marks a name as page junk rather than a +// person — decorative / UI / asset captions that pass NAME_RE. +const HARD_JUNK_WORDS = new Set([ + "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", + "thumbnail", "placeholder", "gallery", "slideshow", "carousel", "homepage", + "webpage", "sidebar", "footer", "header", "menu", "map", "wayback", + "machine", +]); +// Exact lowercase phrases observed polluting the People list. +const KNOWN_JUNK_PHRASES = new Set([ + "updated homepage image", "our mission", "map of the money", + "wayback machine", "deep tech", "startup mentorship hub", "read more", + "learn more", "our team", "the team", "our story", "our values", + "our vision", "get started", "coming soon", "page not found", +]); +// Exact lowercase place / region names that get scraped as "people". +const PLACE_NAMES = new Set([ + "north america", "south america", "central america", "latin america", + "united states", "united kingdom", "middle east", "european union", + "north", "south", "east", "west", "europe", "asia", "africa", "oceania", + "antarctica", "americas", "global", "worldwide", +]); +function normToken(t) { + return t.toLowerCase().normalize("NFKD").replace(/[^a-z0-9]/g, ""); +} +function orgRoleForTokens(tokensLower) { + const set = new Set(tokensLower); + if (set.has("accelerator") || set.has("incubator")) + return "accelerator"; + if (set.has("fund") || set.has("funds")) + return "fund"; + for (const t of ["capital", "ventures", "venture", "partners", "partner", + "equity", "management", "advisors", "associates", "holdings", + "securities", "syndicate"]) { + if (set.has(t)) + return "investor_firm"; + } + return "firm"; +} +/** Map an inferred org role to the `firms.kind` taxonomy for dual-write. */ +export function orgRoleToFirmKind(role) { + switch (role) { + case "accelerator": return "accelerator"; + case "fund": return "fund"; + case "investor_firm": return "vc"; + default: return null; + } +} +/** + * Pure classifier for a name recorded as a `person`. Conservative by + * design: only returns `organization` / `junk` when the signal is clear, + * otherwise `person` (a plausible human name) or `uncertain`. Callers + * decide what to DO with each verdict (reclassify, soft-delete, review). + */ +export function classifyPersonName(rawName) { + const raw = (rawName ?? "").trim(); + if (!raw) + return { verdict: "junk", reasons: ["empty_name"] }; + const lower = raw.toLowerCase(); + const tokens = raw.split(/\s+/).filter(Boolean); + const norm = tokens.map(normToken).filter(Boolean); + // 1. Exact known-junk phrase / place name. + if (KNOWN_JUNK_PHRASES.has(lower)) + return { verdict: "junk", reasons: ["known_junk_phrase"] }; + if (PLACE_NAMES.has(lower)) + return { verdict: "junk", reasons: ["place_name"] }; + // 2. Hard junk word present (image / logo / homepage / map / ...). + // PRECISION GUARD (precision-over-recall): a single junk token inside an + // otherwise-clean two-token Title-Case name ("John Banner", "John Map") + // must NOT auto-delete a plausible real person. Real decorative captions + // are ≥3 tokens ("Updated Homepage Image") or all-generic two-token + // phrases ("Wayback Machine", caught by rule 4 below), so deferring the + // junk-word rule for clean two-token names keeps every junk fixture while + // protecting people whose surname happens to collide with an asset word. + const cleanTwoToken = tokens.length === 2 && tokens.every((t) => /^[\p{Lu}][\p{L}'’.\-]*$/u.test(t)); + if (!cleanTwoToken) { + for (const t of norm) { + if (HARD_JUNK_WORDS.has(t)) + return { verdict: "junk", reasons: [`junk_word:${t}`] }; + } + } + // 3. Organization-suffix detection. + const last = norm[norm.length - 1] ?? ""; + const hasLegal = norm.some((t) => ORG_LEGAL_SUFFIX.has(t)); + const lastIsOrgSuffix = ORG_SUFFIX_LAST.has(last); + if (hasLegal || lastIsOrgSuffix) { + // Need a distinctive (non-generic, non-suffix) token to call it a real + // org. "Intel Capital" → distinctive "intel". "Startup Mentorship Hub" + // → all-generic + suffix → junk. + const distinctive = norm.filter((t, i) => !GENERIC_WORDS.has(t) && + !ORG_SUFFIX_LAST.has(t) && + !ORG_LEGAL_SUFFIX.has(t) && + !(i === norm.length - 1 && lastIsOrgSuffix)); + if (distinctive.length === 0) { + return { verdict: "junk", reasons: ["generic_org_phrase"] }; + } + return { + verdict: "organization", + orgRole: orgRoleForTokens(norm), + reasons: [hasLegal ? "org_legal_suffix" : `org_suffix:${last}`], + }; + } + // 4. Every token is a generic word ("Deep Tech", "Our Mission"). + if (norm.length >= 1 && norm.every((t) => GENERIC_WORDS.has(t))) { + return { verdict: "junk", reasons: ["all_generic_words"] }; + } + // 5. Plausible human name: 2–4 tokens, ≥2 capitalized, not all generic. + const titleTokens = tokens.filter((t) => /^[\p{Lu}]/u.test(t)); + if (tokens.length >= 2 && tokens.length <= 4 && titleTokens.length >= 2) { + return { verdict: "person", reasons: ["plausible_person_name"] }; + } + // 6. Anything else — don't guess. + return { verdict: "uncertain", reasons: ["unclassified"] }; +} +/** Pure detector. NO IO. Safe to call inline on every entity write. */ +export function isGarbage(input) { + const reasons = []; + const raw = (input.display_name ?? "").trim(); + // Rule 1: empty or whitespace-only name. + if (!raw) { + reasons.push("empty_name"); + return { is_garbage: true, reasons }; + } + // Rule 2: name longer than 80 chars. + if (raw.length > NAME_MAX_LEN) + reasons.push("name_too_long"); + // Rule 3: pure emoji / icon (no letters or digits). + if (NO_ALNUM_RE.test(raw)) + reasons.push("no_alphanumeric_chars"); + // Rule 4: page-title with `|` brand fragment. + if (PIPE_TITLE_RE.test(raw)) + reasons.push("page_title_pipe_fragment"); + // Rule 5: blog/press leader phrase. + if (LEADER_RE.test(raw)) + reasons.push("press_leader_phrase"); + // Rule 6: known UI / nav string (case-insensitive exact match). + if (KNOWN_UI_STRINGS.has(raw.toLowerCase())) + reasons.push("known_ui_string"); + // Rule 6c: listicle / directory page title captured as an entity name. + if (LISTICLE_RES.some((re) => re.test(raw))) + reasons.push("listicle_page_title"); + // Rule 6d: name is just the domain slug — the crawler never extracted a + // real name and fell back to the URL. + if (looksDomainDerived(raw, input.primary_domain)) + reasons.push("domain_slug_name"); + // Rule 6b (Task #6 Section A/F): literal HTML entity in name + // (e.g. "Founder & Partner", "Acme & Co"). These are parser + // bugs upstream — the entity should have been decoded before write. + // We flag them here as garbage so they soft-delete on the next sweep + // and surface to the operator console; the durable fix is at the + // scraper layer via decodeEntities() (Task #6 Section F). + if (/&(amp|lt|gt|quot|#x?[0-9a-f]+);/i.test(raw)) + reasons.push("literal_html_entity"); + // Rule 7: person-specific constraints — must contain a space AND + // must not contain pipe / slash / colon. Real human display names + // are "First Last", not "Contact | Sequoia" or "team/people:1". + if (input.kind === "person") { + if (!/\s/.test(raw)) + reasons.push("person_no_space"); + if (/[|/:]/.test(raw)) + reasons.push("person_contains_separator"); + // Task #6: generic page-junk names recorded as people ("Updated + // Homepage Image", "Our Mission", "North America"). Organization + // names are NOT flagged here — they're reclassified (not deleted) + // by the createEntity write guard and the sweep. + const cls = classifyPersonName(raw); + if (cls.verdict === "junk") { + for (const code of cls.reasons) + reasons.push(`name_${code}`); + } + } + return { is_garbage: reasons.length > 0, reasons }; +} +// --------------------------------------------------------------------------- +// Structural rule (requires DB lookups): zero facts AND zero relationships +// AND zero contact channels AND crawler-created >24h ago. Used by the +// cron sweep — NOT by the pre-insert guard (the entity hasn't been +// written yet, so it has no joins). +// --------------------------------------------------------------------------- +export async function isStructurallyOrphan(env, entityId, options = {}) { + const minAge = options.minAgeHours ?? 24; + const reasons = []; + try { + const row = await env.DB.prepare(`SELECT + (SELECT COUNT(*) FROM facts WHERE entity_id = ?1) AS facts, + (SELECT COUNT(*) FROM rel_edges WHERE src_entity_id = ?1 OR dst_entity_id = ?1) AS rels, + (SELECT COUNT(*) FROM channels WHERE entity_id = ?1) AS chans, + (SELECT (julianday('now') - julianday(created_at)) * 24 FROM u_entities WHERE id = ?1) AS age_hours`).bind(entityId).first(); + if (!row) + return { orphan: false, reasons }; + const ageHours = Number(row.age_hours ?? 0); + if (Number(row.facts) === 0 && Number(row.rels) === 0 && Number(row.chans) === 0 && ageHours >= minAge) { + reasons.push("structural_orphan_no_signal"); + return { orphan: true, reasons }; + } + } + catch (e) { + // Optional source tables (channels) may be missing in test + // DBs — degrade to "not orphan" rather than throwing. Per the + // Task #14 honest-degradation pattern. + console.warn("isStructurallyOrphan probe failed", entityId, e.message); + } + return { orphan: false, reasons }; +} +const AI_PROMPT = `You are a data-quality filter for a CRM. Given a candidate \ +entity record, decide whether the display_name is a real person/organization \ +name or noise scraped from an HTML page (page titles, nav labels, "Contact Us", \ +press headlines like "Introducing X", marketing blurbs, etc.). +Reply ONLY as compact JSON: {"verdict":"garbage|real|uncertain","confidence":0.0-1.0,"reason":""}.`; +export async function aiSecondOpinion(env, input) { + if (!env.AI || typeof env.AI.run !== "function") { + return { verdict: "uncertain", confidence: 0, reason: "ai_binding_missing" }; + } + const payload = { + kind: input.kind, + display_name: input.display_name ?? null, + primary_url: input.primary_url ?? null, + primary_domain: input.primary_domain ?? null, + }; + try { + const res = (await env.AI.run("@cf/meta/llama-3.1-8b-instruct-fast", { + messages: [ + { role: "system", content: AI_PROMPT }, + { role: "user", content: JSON.stringify(payload) }, + ], + max_tokens: 80, + })); + const text = typeof res === "string" ? res : (res?.response ?? ""); + const match = text.match(/\{[\s\S]*\}/); + if (!match) + return { verdict: "uncertain", confidence: 0, reason: "ai_no_json" }; + const parsed = JSON.parse(match[0]); + const verdict = parsed.verdict === "garbage" || parsed.verdict === "real" ? parsed.verdict : "uncertain"; + const conf = typeof parsed.confidence === "number" && Number.isFinite(parsed.confidence) + ? Math.max(0, Math.min(1, parsed.confidence)) + : 0; + return { verdict, confidence: conf, reason: typeof parsed.reason === "string" ? parsed.reason : undefined }; + } + catch (e) { + return { verdict: "uncertain", confidence: 0, reason: "ai_error:" + e.message }; + } +} +/** + * Combined verdict: heuristic detector + optional AI second opinion + * for names 30–60 chars that don't match any heuristic. AI flags only + * when verdict='garbage' AND confidence > 0.8. When AI is unavailable + * or returns 'uncertain', the entity is NOT flagged. + */ +export async function evaluateEntity(env, input, opts = {}) { + const heur = isGarbage(input); + if (heur.is_garbage) + return heur; + if (opts.skipAi) + return heur; + const name = (input.display_name ?? "").trim(); + if (name.length < 30 || name.length > 60) + return heur; + const ai = await aiSecondOpinion(env, input); + if (ai.verdict === "garbage" && ai.confidence > 0.8) { + return { is_garbage: true, reasons: ["ai_second_opinion", `ai_conf:${ai.confidence.toFixed(2)}`] }; + } + return heur; +} +// --------------------------------------------------------------------------- +// Soft-delete / restore / purge helpers. All write through +// `data_quality_log` so the operator console can audit every transition. +// --------------------------------------------------------------------------- +export async function logDataQuality(env, entityId, issue, reasons, source, actorEmail, +/** Stored in reasons_json instead of `reasons` when given. Used to park a + * structured snapshot (see softDeleteEntity) rather than a reason list. */ +payload) { + try { + await env.DB.prepare(`INSERT INTO data_quality_log (entity_id, issue, reasons_json, source, actor_email) + VALUES (?, ?, ?, ?, ?)`).bind(entityId, issue, JSON.stringify(payload ?? reasons), source, actorEmail ?? null).run(); + } + catch (e) { + console.warn("data_quality_log insert failed", entityId, e.message); + } +} +export async function softDeleteEntity(env, entityId, reasons, source, actorEmail) { + await env.DB.prepare(`UPDATE u_entities + SET status = 'soft_deleted', + deleted_reason = COALESCE(deleted_reason, ?), + updated_at = datetime('now') + WHERE id = ? AND status != 'soft_deleted'`).bind("garbage_detector_v1:" + reasons.join(","), entityId).run(); + try { + // Park the roles before deleting them. Without this the delete is a + // one-way door: restoreEntity flips status back to 'active' and nothing + // puts the roles back, so a restored entity returns with no roles at all + // and stays invisible on every role-filtered surface — the investor + // lists, the persona matchers, the founder screens. That makes + // "quarantine, then restore the false positives" a promise the code + // cannot keep, which is the whole point of soft delete over purge. + const roles = await env.DB.prepare(`SELECT role, is_primary, source, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(); + // Parked unconditionally, including the empty case. If the park were + // written only when roles exist, an entity that was restored and later + // soft-deleted again with no roles would leave the first park as the + // newest one, and the second restore would replay roles the entity no + // longer had. One row per soft-delete keeps "newest park" and "newest + // soft-delete" the same event. + await logDataQuality(env, entityId, "soft_deleted_roles", [], source, actorEmail, roles.results ?? []); + await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entityId).run(); + } + catch (e) { + console.warn("entity_roles delete during soft-delete failed", entityId, e.message); + } + await logDataQuality(env, entityId, "soft_deleted", reasons, source, actorEmail); +} +export async function restoreEntity(env, entityId, actorEmail) { + await env.DB.prepare(`UPDATE u_entities + SET status = 'active', + deleted_reason = NULL, + updated_at = datetime('now') + WHERE id = ?`).bind(entityId).run(); + // Put back the roles the soft delete removed. Restoring status alone + // returned an entity with no roles, so it never reappeared on any + // role-filtered surface — the restore looked like it worked and did not. + let rolesRestored = 0; + try { + const parked = await env.DB.prepare(`SELECT reasons_json FROM data_quality_log + WHERE entity_id = ? AND issue = 'soft_deleted_roles' + ORDER BY detected_at DESC, id DESC LIMIT 1`).bind(entityId).first(); + const rows = parked?.reasons_json ? JSON.parse(parked.reasons_json) : []; + for (const r of Array.isArray(rows) ? rows : []) { + if (!r || typeof r.role !== "string" || !r.role) + continue; + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, role) DO NOTHING`).bind(entityId, r.role, r.is_primary ?? 0, r.source ?? "restore", r.confidence ?? 0.5).run(); + rolesRestored += 1; + } + } + catch (e) { + // A restore that cannot replay the roles is still better than none — + // but it must not look clean, so it is recorded rather than swallowed. + await logDataQuality(env, entityId, "restore_roles_failed", [e.message], "operator", actorEmail); + } + await logDataQuality(env, entityId, "restored", [`roles_restored:${rolesRestored}`], "operator", actorEmail); +} +export async function purgeEntity(env, entityId, actorEmail) { + // Best-effort cascade across the optional referencing tables; each + // wrapped in its own try/catch so a missing table doesn't block the + // primary delete. Per the Task #14 honest-degradation pattern. + const cascades = [ + `DELETE FROM facts WHERE entity_id = ?`, + `DELETE FROM rel_edges WHERE src_entity_id = ? OR dst_entity_id = ?`, + `DELETE FROM channels WHERE entity_id = ?`, + `DELETE FROM entity_roles WHERE entity_id = ?`, + `DELETE FROM entity_history WHERE entity_id = ?`, + `DELETE FROM entity_legacy_map WHERE entity_id = ?`, + ]; + for (const sql of cascades) { + try { + if (sql.includes("OR dst_entity_id")) { + await env.DB.prepare(sql).bind(entityId, entityId).run(); + } + else { + await env.DB.prepare(sql).bind(entityId).run(); + } + } + catch (e) { + // table-missing or FK noise — log and continue + console.warn("purge cascade failed", sql.slice(0, 40), e.message); + } + } + // Log BEFORE the final delete so the audit trail survives even if + // the row-delete races a concurrent reader. data_quality_log keeps + // entity_id as TEXT (no FK), so the row remains queryable. + await logDataQuality(env, entityId, "purged", [], "operator", actorEmail); + await env.DB.prepare(`DELETE FROM u_entities WHERE id = ?`).bind(entityId).run(); +} +// --------------------------------------------------------------------------- +// Task #6: reclassify a person row that is actually an organization. The +// row is FLIPPED in place (kind person→org) so it leaves the People list +// and joins the org world — non-destructive and reversible (the row, its +// facts and relationships are preserved). When a domain/website is known +// we dual-write a `firms` row so it also surfaces in the Firms list; +// upsertFirm's syncFirmToEntity re-resolves the firm to THIS now-org +// entity via the primary_domain match, so no duplicate entity is minted. +// HONEST DEGRADATION: with no domain/website we cannot dedupe a firm row +// (name-only matching mints duplicates), so we skip it and record the gap +// rather than guessing — the entity still leaves People as an org. +// --------------------------------------------------------------------------- +export async function reclassifyPersonAsOrg(env, entity, orgRole, reasons, source, actorEmail) { + // 1. Flip kind in place — removes it from the People list immediately. + await env.DB.prepare(`UPDATE u_entities SET kind = 'org', updated_at = datetime('now') WHERE id = ?`).bind(entity.id).run(); + // 2. Swap person/investor roles for the inferred org role. Capture the + // prior role set into the audit trail FIRST so the reclassification is + // fully reversible: an operator (or a rollback) can restore the original + // roles from the data_quality_log row, not just flip the kind back. + let priorRoles = []; + try { + const existing = await env.DB.prepare(`SELECT role FROM entity_roles WHERE entity_id = ?`).bind(entity.id).all(); + priorRoles = (existing.results ?? []).map((x) => x.role); + await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entity.id).run(); + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, 1, ?, 1) + ON CONFLICT(entity_id, role) DO UPDATE SET is_primary = 1`).bind(entity.id, orgRole, source).run(); + } + catch (e) { + console.warn("reclassify role swap failed", entity.id, e.message); + } + // 3. Dual-write a firms row when we can dedupe by domain/website. + let firmListed = false; + if (entity.display_name && (entity.primary_domain || entity.primary_url)) { + try { + const { upsertFirm } = await import("../scraper/firms_upsert.js"); + await upsertFirm(env, { + name: entity.display_name, + domain: entity.primary_domain ?? null, + website: entity.primary_url ?? null, + kind: orgRoleToFirmKind(orgRole), + }, source); + firmListed = true; + } + catch (e) { + console.warn("reclassify upsertFirm failed", entity.id, e.message); + } + } + await logDataQuality(env, entity.id, "reclassified", [...reasons, `org_role:${orgRole}`, + priorRoles.length ? `prior_roles:${priorRoles.join(",")}` : "prior_roles:none", + firmListed ? "firm_listed" : "firm_row_skipped_no_domain"], source, actorEmail); + return { reclassified: true, firm_listed: firmListed }; +} +export async function runCleanupSweep(env, opts = {}) { + const mode = opts.mode ?? "recent"; + const lookback = opts.lookbackHours ?? 24; + const limit = opts.limit ?? 5000; + const source = opts.source ?? (mode === "all" ? "oneoff_cleanup" : "cron_sweep"); + const where = mode === "all" + ? `status NOT IN ('soft_deleted','merged')` + : `status NOT IN ('soft_deleted','merged') AND created_at >= datetime('now', '-${lookback} hours')`; + const rows = await env.DB.prepare(`SELECT id, kind, display_name, primary_url, primary_domain, primary_email_key, primary_linkedin_key + FROM u_entities + WHERE ${where} + ORDER BY created_at DESC + LIMIT ?`).bind(limit).all(); + const items = rows.results ?? []; + const byReason = {}; + let flagged = 0; + let softDeleted = 0; + let reclassified = 0; + let needsReview = 0; + for (const r of items) { + // Task #6: organization-name disambiguation for `person` rows. + // Orgs scraped as people are RECLASSIFIED into the org world (and the + // Firms list) rather than soft-deleted. A strong personal identifier + // (personal LinkedIn /in/ or an email) contradicting an org-suffix + // name is flagged for operator review instead of auto-flipped — never + // destroy a likely real person. Junk / plausible-person / uncertain + // names fall through to the existing garbage + orphan path below so + // no prior behavior regresses. + if (r.kind === "person") { + const cls = classifyPersonName(r.display_name); + if (cls.verdict === "organization" && cls.orgRole) { + const personalLinkedin = !!r.primary_linkedin_key && /(^|\/)in\//i.test(r.primary_linkedin_key); + const hasEmail = !!r.primary_email_key; + if (personalLinkedin || hasEmail) { + for (const code of cls.reasons) + byReason[`review_${code}`] = (byReason[`review_${code}`] ?? 0) + 1; + await logDataQuality(env, r.id, "needs_review", [...cls.reasons, "org_name_with_person_signal"], source, opts.actorEmail ?? null); + needsReview += 1; + continue; + } + try { + await reclassifyPersonAsOrg(env, { id: r.id, display_name: r.display_name, primary_url: r.primary_url, primary_domain: r.primary_domain }, cls.orgRole, cls.reasons, source, opts.actorEmail ?? null); + reclassified += 1; + byReason[`reclassified_${cls.orgRole}`] = (byReason[`reclassified_${cls.orgRole}`] ?? 0) + 1; + } + catch (e) { + console.warn("sweep reclassify failed", r.id, e.message); + } + continue; + } + } + // Route through evaluateEntity so the AI second opinion fires for + // ambiguous 30–60 char names (when env.AI is bound). Honors + // skipAi for unit tests + operator-requested fast sweeps. + const evald = await evaluateEntity(env, { + kind: r.kind, display_name: r.display_name, + primary_url: r.primary_url, primary_domain: r.primary_domain, + primary_email_key: r.primary_email_key, primary_linkedin_key: r.primary_linkedin_key, + }, { skipAi: opts.skipAi }); + let reasons = evald.reasons; + let flag = evald.is_garbage; + if (!flag) { + // Structural rule — only meaningful when the entity is not + // brand-new (otherwise the crawler may still be writing joins). + const orphan = await isStructurallyOrphan(env, r.id, { minAgeHours: 24 }); + if (orphan.orphan) { + flag = true; + reasons = orphan.reasons; + } + } + if (!flag) + continue; + flagged += 1; + for (const code of reasons) + byReason[code] = (byReason[code] ?? 0) + 1; + try { + await softDeleteEntity(env, r.id, reasons, source, opts.actorEmail ?? null); + softDeleted += 1; + } + catch (e) { + console.warn("sweep soft-delete failed", r.id, e.message); + } + } + const result = { + scanned: items.length, flagged, soft_deleted: softDeleted, + reclassified, needs_review: needsReview, + by_reason: byReason, bounded: items.length >= limit, + }; + console.log("garbage.cleanup_sweep", JSON.stringify({ mode, ...result })); + return result; +} diff --git a/apps/worker/test-dist-q/entities/model.js b/apps/worker/test-dist-q/entities/model.js new file mode 100644 index 00000000..a09d5fab --- /dev/null +++ b/apps/worker/test-dist-q/entities/model.js @@ -0,0 +1,6 @@ +// TS types matching the unified entity DDL (migrations 200–208). +// Task #4: re-export the rich-profile write helpers under a single +// EntityService facade so callers can `import { EntityService } from +// "./entities/model"` for every structured profile write. +export { EntityService } from "./profile"; +export { PREDICATE_REGISTRY, PREDICATE_MAP, EMITTED_PREDICATES, getPredicateMeta, } from "./profile-predicates"; diff --git a/apps/worker/test-dist-q/entities/normalize.js b/apps/worker/test-dist-q/entities/normalize.js new file mode 100644 index 00000000..84318cba --- /dev/null +++ b/apps/worker/test-dist-q/entities/normalize.js @@ -0,0 +1,120 @@ +// Canonical key generators for the unified entity model. Channels are +// keyed by (kind, canonical) and equality must collapse trivial +// presentation differences — case, whitespace, '+suffix' in emails, +// formatting characters in phone numbers, query strings on LinkedIn URLs. +export function canonicalEmail(raw) { + if (!raw) + return null; + const s = String(raw).trim().toLowerCase(); + const m = /^([^@\s]+)@([a-z0-9.-]+\.[a-z]{2,})$/i.exec(s); + if (!m) + return null; + // Strip '+suffix' from the local part (Gmail-style). + const local = m[1].split("+")[0]; + return `${local}@${m[2]}`; +} +export function canonicalPhone(raw) { + if (!raw) + return null; + const digits = String(raw).replace(/[^\d+]/g, ""); + if (!digits) + return null; + // Best-effort E.164: keep leading + if present, else assume already + // includes country code. + return digits.startsWith("+") ? digits : `+${digits.replace(/^\++/, "")}`; +} +export function canonicalLinkedin(raw) { + if (!raw) + return null; + let s = String(raw).trim(); + if (!s) + return null; + // Accept handle-only ("janedoe") or full URL. + if (!/^https?:\/\//i.test(s) && !s.includes("/")) { + return `/in/${s.toLowerCase()}`; + } + try { + const u = new URL(s.startsWith("http") ? s : `https://${s}`); + if (!/linkedin\.com$/i.test(u.hostname) && !/(^|\.)linkedin\.com$/i.test(u.hostname)) + return null; + const p = u.pathname.replace(/\/+$/, "").toLowerCase(); + // /in/ | /company/ | /school/ + const m = /^\/(in|company|school|pub)\/([^/?#]+)/.exec(p); + if (!m) + return null; + return `/${m[1] === "pub" ? "in" : m[1]}/${m[2]}`; + } + catch { + return null; + } +} +export function canonicalTwitter(raw) { + if (!raw) + return null; + const s = String(raw).trim().replace(/^@/, ""); + if (!s) + return null; + if (/^https?:\/\//i.test(s)) { + try { + const u = new URL(s); + const m = /^\/([A-Za-z0-9_]{1,15})\/?$/.exec(u.pathname); + return m ? m[1].toLowerCase() : null; + } + catch { + return null; + } + } + return /^[A-Za-z0-9_]{1,15}$/.test(s) ? s.toLowerCase() : null; +} +export function canonicalGithub(raw) { + if (!raw) + return null; + const s = String(raw).trim().replace(/^@/, ""); + if (!s) + return null; + if (/^https?:\/\//i.test(s)) { + try { + const u = new URL(s); + const m = /^\/([A-Za-z0-9-]{1,39})\/?$/.exec(u.pathname); + return m ? m[1].toLowerCase() : null; + } + catch { + return null; + } + } + return /^[A-Za-z0-9-]{1,39}$/.test(s) ? s.toLowerCase() : null; +} +export function canonicalDomain(raw) { + if (!raw) + return null; + let s = String(raw).trim().toLowerCase(); + if (!s) + return null; + if (/^https?:\/\//.test(s)) { + try { + s = new URL(s).hostname; + } + catch { + return null; + } + } + s = s.replace(/^www\./, "").replace(/\/.*$/, ""); + return /^[a-z0-9.-]+\.[a-z]{2,}$/.test(s) ? s : null; +} +export function canonicalUrl(raw) { + if (!raw) + return null; + try { + const u = new URL(String(raw).trim()); + u.hash = ""; + return u.toString(); + } + catch { + return null; + } +} +export async function sha256(s) { + const buf = new TextEncoder().encode(s); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} diff --git a/apps/worker/test-dist-q/entities/profile-predicates.js b/apps/worker/test-dist-q/entities/profile-predicates.js new file mode 100644 index 00000000..f88842dc --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile-predicates.js @@ -0,0 +1,218 @@ +// Task #4: Predicate registry — single source of truth. +// +// The same list is mirrored into `predicate_registry` by migration 328. +// Every predicate string that `profile.ts` ever passes to `insertFact` +// MUST appear in this array; the CI smoke test (test/profile.test.mjs) +// enforces it. Adding a new predicate is a TWO-file change: +// 1. append it here AND +// 2. add the matching `INSERT OR IGNORE INTO predicate_registry` row in +// migration 328_predicate_registry.sql. +// +// `value_type` is informational metadata for the UI formatter — it does +// not constrain the runtime shape of the fact value_json (those shapes +// live in profile-shapes.ts). +export const PREDICATE_REGISTRY = [ + // ---- identity (rich-profile) ------------------------------------------- + { predicate: "person.identity", label: "Identity snapshot", icon: "user", formatter: "json", category: "identity", value_type: "json", description: "Snapshot of person_identity row." }, + { predicate: "person.identity.full_name", label: "Full name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Legal or commonly used full name." }, + { predicate: "person.identity.preferred_name", label: "Preferred name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "What the person prefers to be called." }, + { predicate: "person.identity.pronouns", label: "Pronouns", icon: "user-circle", formatter: "text", category: "identity", value_type: "json", description: "Pronouns triple (subject/object/possessive)." }, + { predicate: "person.identity.birth_year", label: "Birth year", icon: "cake", formatter: "year", category: "identity", value_type: "year", description: "Year of birth (no full DOB stored)." }, + { predicate: "person.identity.nationality", label: "Nationality", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO-3166-1 alpha-2 nationality." }, + { predicate: "person.identity.languages", label: "Languages", icon: "languages", formatter: "list", category: "identity", value_type: "json", description: "Spoken languages with proficiency." }, + { predicate: "person.identity.timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "IANA tz database name." }, + { predicate: "person.identity.location_city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Current city of residence." }, + { predicate: "person.identity.location_country", label: "Country", icon: "globe", formatter: "flag", category: "identity", value_type: "country_iso2", description: "Current country of residence." }, + { predicate: "person.identity.headshot_url", label: "Headshot", icon: "image", formatter: "avatar", category: "identity", value_type: "url", description: "Public headshot image URL." }, + // ---- career ------------------------------------------------------------- + { predicate: "person.career", label: "Career entry", icon: "briefcase", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a career_history row." }, + // ---- board -------------------------------------------------------------- + { predicate: "person.board_seat", label: "Board seat", icon: "users", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a board_seats row." }, + // ---- education ---------------------------------------------------------- + { predicate: "person.education", label: "Education", icon: "graduation-cap", formatter: "json", category: "education", value_type: "json", description: "Snapshot of an education_history row." }, + // ---- family ------------------------------------------------------------- + { predicate: "person.family_tie", label: "Family tie", icon: "heart", formatter: "json", category: "family", value_type: "json", description: "Snapshot of a family_ties row." }, + // ---- conference --------------------------------------------------------- + { predicate: "person.conference", label: "Conference", icon: "calendar", formatter: "json", category: "conference", value_type: "json", description: "Snapshot of a conference_attendance row." }, + // ---- preferences (one predicate per documented preference_key) --------- + { predicate: "person.preference.communication_channel", label: "Communication channel", icon: "message-circle", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact channel (email/text/dm/voice)." }, + { predicate: "person.preference.contact_time", label: "Best time to reach", icon: "clock", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact window or day." }, + { predicate: "person.preference.meeting_format", label: "Meeting format", icon: "video", formatter: "text", category: "preference", value_type: "text", description: "In-person, video, phone, async." }, + { predicate: "person.preference.gift_dietary", label: "Dietary", icon: "salad", formatter: "text", category: "preference", value_type: "text", description: "Vegan/vegetarian/keto/halal/etc." }, + { predicate: "person.preference.gift_allergies", label: "Allergies", icon: "alert-triangle", formatter: "list", category: "preference", value_type: "json", description: "Food/material allergies to avoid for gifting." }, + { predicate: "person.preference.coffee_order", label: "Coffee order", icon: "coffee", formatter: "text", category: "preference", value_type: "text", description: "Usual coffee order." }, + { predicate: "person.preference.travel_class", label: "Travel class", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Preferred flight cabin class." }, + { predicate: "person.preference.hotel_brand", label: "Hotel brand", icon: "bed", formatter: "text", category: "preference", value_type: "text", description: "Preferred hotel chain or brand." }, + { predicate: "person.preference.airline_status", label: "Airline status", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Frequent-flyer status / preferred carrier." }, + // ---- interests (one per category) -------------------------------------- + { predicate: "person.interest.topic", label: "Topic", icon: "tag", formatter: "badge", category: "interest", value_type: "text", description: "Topic of interest." }, + { predicate: "person.interest.sport", label: "Sport", icon: "trophy", formatter: "badge", category: "interest", value_type: "text", description: "Sport played or followed." }, + { predicate: "person.interest.team", label: "Team", icon: "shield", formatter: "badge", category: "interest", value_type: "text", description: "Favorite team." }, + { predicate: "person.interest.book", label: "Book", icon: "book", formatter: "text", category: "interest", value_type: "text", description: "Book the person recommends or read." }, + { predicate: "person.interest.author", label: "Author", icon: "feather", formatter: "text", category: "interest", value_type: "text", description: "Favorite author." }, + { predicate: "person.interest.podcast", label: "Podcast", icon: "mic", formatter: "text", category: "interest", value_type: "text", description: "Podcast the person listens to." }, + { predicate: "person.interest.music", label: "Music genre", icon: "music", formatter: "badge", category: "interest", value_type: "text", description: "Preferred music genre." }, + { predicate: "person.interest.artist", label: "Artist", icon: "music", formatter: "text", category: "interest", value_type: "text", description: "Favorite musical artist." }, + { predicate: "person.interest.film", label: "Film", icon: "film", formatter: "text", category: "interest", value_type: "text", description: "Favorite film." }, + { predicate: "person.interest.show", label: "TV show", icon: "tv", formatter: "text", category: "interest", value_type: "text", description: "Favorite TV show." }, + { predicate: "person.interest.hobby", label: "Hobby", icon: "puzzle", formatter: "badge", category: "interest", value_type: "text", description: "Hobby outside of work." }, + { predicate: "person.interest.cause", label: "Cause", icon: "heart-handshake", formatter: "badge", category: "interest", value_type: "text", description: "Cause the person supports." }, + // ---- lifestyle signals ------------------------------------------------- + { predicate: "person.lifestyle.runs", label: "Runs", icon: "footprints", formatter: "text", category: "lifestyle", value_type: "json", description: "Is a runner." }, + { predicate: "person.lifestyle.cycles", label: "Cycles", icon: "bike", formatter: "text", category: "lifestyle", value_type: "json", description: "Cycles." }, + { predicate: "person.lifestyle.surfs", label: "Surfs", icon: "waves", formatter: "text", category: "lifestyle", value_type: "json", description: "Surfs." }, + { predicate: "person.lifestyle.skis", label: "Skis", icon: "snowflake", formatter: "text", category: "lifestyle", value_type: "json", description: "Skis or snowboards." }, + { predicate: "person.lifestyle.golfs", label: "Golfs", icon: "flag", formatter: "text", category: "lifestyle", value_type: "json", description: "Plays golf." }, + { predicate: "person.lifestyle.yoga", label: "Yoga", icon: "activity", formatter: "text", category: "lifestyle", value_type: "json", description: "Practices yoga." }, + { predicate: "person.lifestyle.meditates", label: "Meditates", icon: "leaf", formatter: "text", category: "lifestyle", value_type: "json", description: "Meditates regularly." }, + { predicate: "person.lifestyle.cooks", label: "Cooks", icon: "chef-hat", formatter: "text", category: "lifestyle", value_type: "json", description: "Cooks for fun." }, + { predicate: "person.lifestyle.collects", label: "Collects", icon: "package", formatter: "text", category: "lifestyle", value_type: "json", description: "Collects something (art, wine, watches…)." }, + { predicate: "person.lifestyle.pet", label: "Pet", icon: "paw-print", formatter: "text", category: "lifestyle", value_type: "json", description: "Has a pet." }, + { predicate: "person.lifestyle.marathon", label: "Marathon", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed marathon." }, + { predicate: "person.lifestyle.ironman", label: "Ironman", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed Ironman triathlon." }, + // ---- travel ------------------------------------------------------------- + { predicate: "person.travel.frequent_city", label: "Frequent city", icon: "map-pin", formatter: "text", category: "travel", value_type: "text", description: "City the person frequently travels to." }, + { predicate: "person.travel.home_base", label: "Home base", icon: "home", formatter: "text", category: "travel", value_type: "text", description: "Stated home base." }, + { predicate: "person.travel.recent_trip", label: "Recent trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Recent trip place + date window." }, + { predicate: "person.travel.upcoming_trip", label: "Upcoming trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Announced upcoming trip." }, + { predicate: "person.travel.airport_hub", label: "Airport hub", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Home airport (IATA code)." }, + // ---- goals -------------------------------------------------------------- + { predicate: "person.goal.short_term", label: "Short-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for <12 months." }, + { predicate: "person.goal.long_term", label: "Long-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for 12+ months." }, + { predicate: "person.goal.hiring", label: "Hiring goal", icon: "user-plus", formatter: "text", category: "goal", value_type: "text", description: "Stated hiring need." }, + { predicate: "person.goal.fundraising", label: "Fundraising goal", icon: "dollar-sign", formatter: "text", category: "goal", value_type: "text", description: "Stated fundraising plan." }, + { predicate: "person.goal.investing_thesis", label: "Investing thesis", icon: "lightbulb", formatter: "text", category: "goal", value_type: "text", description: "Stated investing thesis." }, + { predicate: "person.goal.expansion_market", label: "Expansion market", icon: "map", formatter: "text", category: "goal", value_type: "text", description: "Stated market expansion." }, + // ---- conversation hooks ------------------------------------------------ + { predicate: "person.hook.recent_news", label: "Recent news", icon: "newspaper", formatter: "text", category: "hook", value_type: "text", description: "Recent news about the person." }, + { predicate: "person.hook.shared_connection", label: "Shared connection", icon: "users", formatter: "text", category: "hook", value_type: "text", description: "Mutual contact you can mention." }, + { predicate: "person.hook.shared_school", label: "Shared school", icon: "graduation-cap", formatter: "text", category: "hook", value_type: "text", description: "Shared alma mater." }, + { predicate: "person.hook.shared_employer", label: "Shared employer", icon: "briefcase", formatter: "text", category: "hook", value_type: "text", description: "Shared former or current employer." }, + { predicate: "person.hook.shared_interest", label: "Shared interest", icon: "tag", formatter: "text", category: "hook", value_type: "text", description: "Shared interest or hobby." }, + { predicate: "person.hook.recent_post", label: "Recent post", icon: "message-square", formatter: "text", category: "hook", value_type: "text", description: "Recent public post by the person." }, + { predicate: "person.hook.life_event", label: "Life event", icon: "sparkles", formatter: "text", category: "hook", value_type: "text", description: "Life event (job change, baby, move)." }, + { predicate: "person.hook.opinion_quoted", label: "Opinion quoted", icon: "quote", formatter: "text", category: "hook", value_type: "text", description: "Public statement on a topic." }, + // ---- appreciation ------------------------------------------------------ + { predicate: "person.appreciation.compliment_topic", label: "Compliment topic", icon: "thumbs-up", formatter: "text", category: "appreciation", value_type: "text", description: "Topic the person likes to be complimented on." }, + { predicate: "person.appreciation.gift_idea", label: "Gift idea", icon: "gift", formatter: "text", category: "appreciation", value_type: "text", description: "Concrete gift idea." }, + { predicate: "person.appreciation.charity_supported", label: "Charity supported", icon: "heart", formatter: "text", category: "appreciation", value_type: "text", description: "Charity the person supports." }, + { predicate: "person.appreciation.cause_advocated", label: "Cause advocated", icon: "megaphone", formatter: "text", category: "appreciation", value_type: "text", description: "Cause the person publicly advocates." }, + { predicate: "person.appreciation.recognition_received", label: "Recognition received", icon: "award", formatter: "text", category: "appreciation", value_type: "text", description: "Public award/recognition received." }, + // ---- legacy / extractor predicates (so the UI never renders a raw key) - + { predicate: "name", label: "Name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Display name (extractor field)." }, + { predicate: "title", label: "Title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Job title (extractor field)." }, + { predicate: "role", label: "Role", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Functional role." }, + { predicate: "employer", label: "Employer", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Current employer name." }, + { predicate: "company", label: "Company", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Alias for employer in some extractors." }, + { predicate: "headline", label: "Headline", icon: "type", formatter: "text", category: "identity", value_type: "text", description: "LinkedIn-style headline." }, + { predicate: "summary", label: "Summary", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Bio / summary paragraph." }, + { predicate: "bio", label: "Bio", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Profile bio." }, + { predicate: "description", label: "Description", icon: "file-text", formatter: "text", category: "firm", value_type: "text", description: "Org/company description." }, + { predicate: "location", label: "Location", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Stated location string." }, + { predicate: "city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "City (extractor field)." }, + { predicate: "region", label: "Region", icon: "map", formatter: "text", category: "identity", value_type: "text", description: "State/region." }, + { predicate: "country", label: "Country", icon: "globe", formatter: "text", category: "identity", value_type: "text", description: "Country (name string)." }, + { predicate: "country_iso2", label: "Country (ISO)", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO 3166-1 alpha-2 country." }, + { predicate: "timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "Timezone (extractor field)." }, + { predicate: "email", label: "Email", icon: "mail", formatter: "link", category: "contact", value_type: "email", description: "Email address." }, + { predicate: "phone", label: "Phone", icon: "phone", formatter: "text", category: "contact", value_type: "text", description: "Phone number (E.164 preferred)." }, + { predicate: "linkedin_url", label: "LinkedIn", icon: "linkedin", formatter: "link", category: "contact", value_type: "url", description: "LinkedIn profile URL." }, + { predicate: "twitter_url", label: "Twitter", icon: "twitter", formatter: "link", category: "contact", value_type: "url", description: "Twitter/X profile URL." }, + { predicate: "twitter_handle", label: "Twitter handle", icon: "twitter", formatter: "text", category: "contact", value_type: "text", description: "Twitter/X handle." }, + { predicate: "github_url", label: "GitHub", icon: "github", formatter: "link", category: "contact", value_type: "url", description: "GitHub profile URL." }, + { predicate: "github_handle", label: "GitHub handle", icon: "github", formatter: "text", category: "contact", value_type: "text", description: "GitHub handle." }, + { predicate: "website", label: "Website", icon: "link", formatter: "link", category: "contact", value_type: "url", description: "Personal/firm website URL." }, + { predicate: "primary_domain", label: "Primary domain", icon: "globe", formatter: "text", category: "firm", value_type: "text", description: "Canonical apex domain." }, + { predicate: "sector", label: "Sector", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Sector / vertical tag." }, + { predicate: "stage", label: "Stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Investment stage focus." }, + { predicate: "check_size_min_usd", label: "Min check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Minimum typical check size." }, + { predicate: "check_size_max_usd", label: "Max check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Maximum typical check size." }, + { predicate: "fund_size_usd", label: "Fund size (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent fund size." }, + { predicate: "founded_year", label: "Founded year", icon: "calendar", formatter: "year", category: "firm", value_type: "year", description: "Year founded." }, + { predicate: "founded_at", label: "Founded", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Founding date (full)." }, + { predicate: "funding_stage", label: "Funding stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Latest funding stage." }, + { predicate: "total_funding_usd", label: "Total funding (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Total funding raised." }, + { predicate: "last_round_usd", label: "Last round (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent round amount." }, + { predicate: "hq_city", label: "HQ city", icon: "map-pin", formatter: "text", category: "firm", value_type: "text", description: "HQ city." }, + { predicate: "hq_country_iso2", label: "HQ country (ISO)", icon: "flag", formatter: "flag", category: "firm", value_type: "country_iso2", description: "HQ country ISO code." }, + { predicate: "employees", label: "Employees", icon: "users", formatter: "text", category: "firm", value_type: "number", description: "Employee headcount or band." }, + { predicate: "industry", label: "Industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry classification." }, + { predicate: "display_name", label: "Display name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Canonical display name." }, + // ---- Task #1: SEC EDGAR deep-adapter predicates ----------------------- + // Mirror in migration 349_sec_edgar.sql — two-file change enforced by + // test/profile.test.mjs. + { predicate: "sec.cik", label: "SEC CIK", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC Central Index Key (10-digit, zero-padded)." }, + { predicate: "sec.crd", label: "SEC CRD", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Investment Adviser CRD# from Form ADV." }, + { predicate: "sec.cusip", label: "CUSIP", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Committee on Uniform Securities Identification Procedures code." }, + { predicate: "sec.ticker", label: "Ticker", icon: "trending-up", formatter: "badge", category: "identity", value_type: "text", description: "Public ticker symbol." }, + { predicate: "sec.sec_file_number", label: "SEC file number", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC file number (e.g. 801-12345 for advisers)." }, + { predicate: "sec.fund_id_807", label: "SEC fund ID", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC fund identifier (807-XXXXXXXX)." }, + { predicate: "aum_usd", label: "AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Assets under management in USD." }, + { predicate: "sec.form_adv.filed_at", label: "Last Form ADV filed", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Most recent Form ADV acceptance date." }, + { predicate: "sec.form_adv.fund", label: "Form ADV fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Fund disclosed on Schedule D §7.B.(1)." }, + { predicate: "sec.form_d.round", label: "Form D round", icon: "dollar-sign", formatter: "json", category: "firm", value_type: "json", description: "Private placement disclosed on Form D." }, + { predicate: "sec.form_d.issuer_industry", label: "Form D industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry group declared on Form D." }, + { predicate: "sec.13f.holding", label: "13F holding", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Equity position disclosed on Form 13F-HR." }, + { predicate: "sec.13f.filer_aum_usd", label: "13F filer AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Aggregate USD value of 13F holdings (proxy AUM)." }, + { predicate: "sec.13d.beneficial_owner", label: "13D beneficial owner", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Schedule 13D 5%+ beneficial-ownership disclosure." }, + { predicate: "sec.form4.insider_trade", label: "Insider trade", icon: "arrow-up-down", formatter: "json", category: "firm", value_type: "json", description: "Form 4 §16 insider transaction." }, + { predicate: "sec.form4.officer_title", label: "Officer title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Officer title declared on Form 4 (when reporter is officer)." }, + { predicate: "sec.s1.ipo_intent", label: "S-1 IPO intent", icon: "rocket", formatter: "text", category: "firm", value_type: "text", description: "Company filed Form S-1 (IPO registration)." }, + { predicate: "sec.s1.underwriter", label: "IPO underwriter", icon: "briefcase", formatter: "text", category: "firm", value_type: "text", description: "Underwriter listed on Form S-1." }, + { predicate: "sec.8k.material_event", label: "8-K material event", icon: "alert-circle", formatter: "json", category: "firm", value_type: "json", description: "Form 8-K current report item." }, + { predicate: "sec.10k.revenue_usd", label: "10-K revenue (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Annual revenue from Form 10-K." }, + { predicate: "sec.10k.net_income_usd", label: "10-K net income", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Net income from Form 10-K." }, + { predicate: "sec.10k.fiscal_year_end", label: "10-K fiscal year-end", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Fiscal year-end from Form 10-K." }, + { predicate: "sec.10k.executive", label: "10-K executive", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Named executive officer compensation from Form 10-K." }, + { predicate: "sec.pf.fund", label: "Form PF fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Private fund disclosure from Form PF (large private fund adviser)." }, + { predicate: "sec.gp_disclosed", label: "GP disclosed (SEC)", icon: "user-check", formatter: "text", category: "firm", value_type: "text", description: "GP / control-person disclosed on a SEC filing." }, +]; +export const PREDICATE_MAP = Object.freeze(Object.fromEntries(PREDICATE_REGISTRY.map((p) => [p.predicate, p]))); +export function getPredicateMeta(predicate) { + return PREDICATE_MAP[predicate] ?? null; +} +// Helpers in `profile.ts` MUST only emit predicates listed here. +// The smoke test asserts EMITTED_PREDICATES ⊆ PREDICATE_REGISTRY keys. +export const EMITTED_PREDICATES = Object.freeze([ + // static (one per helper) + "person.identity", + "person.career", + "person.board_seat", + "person.education", + "person.family_tie", + "person.conference", + // dynamic: person.preference.{key} + "person.preference.communication_channel", + "person.preference.contact_time", + "person.preference.meeting_format", + "person.preference.gift_dietary", + "person.preference.gift_allergies", + "person.preference.coffee_order", + "person.preference.travel_class", + "person.preference.hotel_brand", + "person.preference.airline_status", + // dynamic: person.interest.{category} + "person.interest.topic", "person.interest.sport", "person.interest.team", + "person.interest.book", "person.interest.author", "person.interest.podcast", + "person.interest.music", "person.interest.artist", "person.interest.film", + "person.interest.show", "person.interest.hobby", "person.interest.cause", + // dynamic: person.lifestyle.{key} + "person.lifestyle.runs", "person.lifestyle.cycles", "person.lifestyle.surfs", + "person.lifestyle.skis", "person.lifestyle.golfs", "person.lifestyle.yoga", + "person.lifestyle.meditates", "person.lifestyle.cooks", "person.lifestyle.collects", + "person.lifestyle.pet", "person.lifestyle.marathon", "person.lifestyle.ironman", + // dynamic: person.travel.{kind} + "person.travel.frequent_city", "person.travel.home_base", + "person.travel.recent_trip", "person.travel.upcoming_trip", "person.travel.airport_hub", + // dynamic: person.goal.{kind} + "person.goal.short_term", "person.goal.long_term", "person.goal.hiring", + "person.goal.fundraising", "person.goal.investing_thesis", "person.goal.expansion_market", + // dynamic: person.hook.{kind} + "person.hook.recent_news", "person.hook.shared_connection", "person.hook.shared_school", + "person.hook.shared_employer", "person.hook.shared_interest", "person.hook.recent_post", + "person.hook.life_event", "person.hook.opinion_quoted", + // dynamic: person.appreciation.{kind} + "person.appreciation.compliment_topic", "person.appreciation.gift_idea", + "person.appreciation.charity_supported", "person.appreciation.cause_advocated", + "person.appreciation.recognition_received", +]); diff --git a/apps/worker/test-dist-q/entities/profile-shapes.js b/apps/worker/test-dist-q/entities/profile-shapes.js new file mode 100644 index 00000000..ab1d4f7f --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile-shapes.js @@ -0,0 +1,4 @@ +// Task #4: Typed shapes for the JSON columns on rich-profile tables. +// Every helper in `profile.ts` serializes through these shapes; nothing +// passes raw `unknown` through `JSON.stringify`. +export {}; diff --git a/apps/worker/test-dist-q/entities/profile.js b/apps/worker/test-dist-q/entities/profile.js new file mode 100644 index 00000000..7de9b64d --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile.js @@ -0,0 +1,681 @@ +// Task #4: EntityService write helpers for the rich person profile. +// +// Every helper: +// 1. Validates input against the typed shape (profile-shapes.ts). +// 2. Acquires the per-entity DO lock (EntityLock /acquire) so concurrent +// scrapers / OSINT / agent calls don't race on the same entity_id. +// 3. Upserts the structured row using a stable natural key +// (ON CONFLICT(...) DO UPDATE). +// 4. Mirrors a canonical row into `facts` via insertFact — the fact's +// hash dedupe key is sha256(entity|predicate|value|source_url) so the +// second call with identical content updates `observed_at` rather +// than creating a duplicate row. +// +// Public-signal constraint: every helper except `setPersonIdentity` (which +// allows operator-asserted rows with isOperatorAsserted=true) refuses to +// write without a source_url. +import { sha256 } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +import { EMITTED_PREDICATES, PREDICATE_MAP, } from "./profile-predicates"; +// ---- Internal: per-entity lock (token mutex via EntityLock DO) ----------- +// +// Mirrors the OSINT resolver's acquire/release pattern so all rich-profile +// writes for a given entity_id serialize against OSINT, dual-write, and +// any other caller that already uses the same DO id-namespace. +// +// Strict serialization (task contract): when the ENTITY_LOCK binding is +// present we MUST hold the lock for the duration of the write. If the +// DO is unreachable or the lock is currently held by another writer we +// retry with exponential backoff and then fail loudly rather than +// silently racing. The only bypass is when the binding itself is absent +// — that's how the in-memory unit tests run, and it's explicit. +export class ProfileLockError extends Error { + constructor(entityId, reason) { + super(`profile lock for ${entityId}: ${reason}`); + this.name = "ProfileLockError"; + } +} +async function withProfileLock(env, entityId, fn) { + if (!env.ENTITY_LOCK) + return await fn(); + const stub = env.ENTITY_LOCK.get(env.ENTITY_LOCK.idFromName(entityId)); + const token = crypto.randomUUID(); + let acquired = false; + let lastErr = "unknown"; + // 5 attempts, 50 / 100 / 200 / 400 / 800ms backoff = ≤1.55s total. + // Beyond that we surface the error so the caller can decide (retry the + // job, surface to the operator, etc.) rather than corrupt with a + // racy write. + for (let attempt = 0; attempt < 5 && !acquired; attempt++) { + if (attempt > 0) { + await new Promise((r) => setTimeout(r, 50 * 2 ** (attempt - 1))); + } + try { + const res = await stub.fetch("https://lock/acquire", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ token, ttlMs: 60_000 }), + }); + if (res.ok) { + acquired = true; + break; + } + lastErr = `acquire returned ${res.status}`; + } + catch (e) { + lastErr = e.message || "fetch failed"; + } + } + if (!acquired) { + throw new ProfileLockError(entityId, lastErr); + } + try { + return await fn(); + } + finally { + try { + await stub.fetch("https://lock/release", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ token }), + }); + } + catch { /* lock will TTL out after 60s */ } + } +} +// ---- Internal: mirror a structured row into `facts` --------------------- +// +// Centralized projection so every helper produces consistent fact rows. +// +// Dedupe contract (task spec): natural key is (entity_id, predicate, +// source_url) — independent of value. Re-observing the same predicate +// from the same source_url MUST upsert the existing fact (new value, +// refreshed observed_at) rather than create a second row. The hash we +// compute here covers ONLY those three fields, and UNIQUE(hash) on +// `facts` (migration 201) turns a collision into our UPDATE path. +// +// We bypass `insertFact` for the mirror path because its hash includes +// the value (which is correct for raw scraper writes but wrong here — +// it would let value drift create duplicate (entity, predicate, source) +// rows). The summary-rebuild enqueue is still triggered. +// +// Asserts the predicate exists in the registry — catches typos at +// runtime and is defense-in-depth backup to the smoke-test enforcement +// of EMITTED_PREDICATES ⊆ registry. +async function mirrorFact(env, args) { + if (!PREDICATE_MAP[args.predicate]) { + throw new Error(`profile.mirrorFact: predicate "${args.predicate}" is not in PREDICATE_REGISTRY`); + } + const hash = await sha256(`${args.entityId}|${args.predicate}|${args.sourceUrl}`); + const valueText = args.valueText ?? null; + const valueNumber = args.valueNumber ?? null; + const valueJsonStr = args.valueJson != null ? JSON.stringify(args.valueJson) : null; + const confidence = args.confidence ?? 1.0; + const now = args.observedAt ?? new Date().toISOString(); + try { + await env.DB.prepare(`INSERT INTO facts ( + id, entity_id, predicate, value_text, value_number, value_json, + value_entity_id, source_kind, source, evidence_url, confidence, + observed_at, valid_from, valid_to, is_current, hash + ) VALUES (?, ?, ?, ?, ?, ?, NULL, 'enrichment', ?, ?, ?, ?, NULL, NULL, 1, ?)`).bind(crypto.randomUUID(), args.entityId, args.predicate, valueText, valueNumber, valueJsonStr, args.sourceUrl, args.sourceUrl, confidence, now, hash).run(); + } + catch (e) { + const msg = e.message || ""; + if (/UNIQUE/i.test(msg)) { + await env.DB.prepare(`UPDATE facts + SET value_text = ?, value_number = ?, value_json = ?, + confidence = MAX(confidence, ?), + observed_at = ?, is_current = 1 + WHERE hash = ?`).bind(valueText, valueNumber, valueJsonStr, confidence, now, hash).run(); + } + else { + throw e; + } + } + try { + await enqueueSummaryRebuild(env, args.entityId); + } + catch { /* best-effort */ } +} +function requireSourceUrl(helper, url) { + if (!url || typeof url !== "string" || url.trim().length === 0) { + throw new Error(`profile.${helper}: source_url is required (public-signal-only constraint)`); + } + return url; +} +function requireNonEmpty(helper, field, v) { + if (!v || typeof v !== "string" || v.trim().length === 0) { + throw new Error(`profile.${helper}: ${field} is required`); + } + return v.trim(); +} +function nowIso() { return new Date().toISOString(); } +// ========================================================================= +// 1. setPersonIdentity — upsert on entity_id. +// ========================================================================= +export async function setPersonIdentity(env, input) { + requireNonEmpty("setPersonIdentity", "entityId", input.entityId); + const isOperator = input.isOperatorAsserted === true; + if (!isOperator) + requireSourceUrl("setPersonIdentity", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_identity ( + entity_id, full_name, preferred_name, pronouns_json, birth_year, + nationality, languages_json, timezone, location_city, location_country, + headshot_url, source_url, is_operator_asserted, confidence, + observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id) DO UPDATE SET + full_name = COALESCE(excluded.full_name, person_identity.full_name), + preferred_name = COALESCE(excluded.preferred_name, person_identity.preferred_name), + pronouns_json = COALESCE(excluded.pronouns_json, person_identity.pronouns_json), + birth_year = COALESCE(excluded.birth_year, person_identity.birth_year), + nationality = COALESCE(excluded.nationality, person_identity.nationality), + languages_json = COALESCE(excluded.languages_json, person_identity.languages_json), + timezone = COALESCE(excluded.timezone, person_identity.timezone), + location_city = COALESCE(excluded.location_city, person_identity.location_city), + location_country = COALESCE(excluded.location_country, person_identity.location_country), + headshot_url = COALESCE(excluded.headshot_url, person_identity.headshot_url), + source_url = COALESCE(excluded.source_url, person_identity.source_url), + is_operator_asserted = MAX(excluded.is_operator_asserted, person_identity.is_operator_asserted), + confidence = MAX(excluded.confidence, person_identity.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(input.entityId, input.fullName ?? null, input.preferredName ?? null, input.pronouns ? JSON.stringify(input.pronouns) : null, input.birthYear ?? null, input.nationality ?? null, input.languages ? JSON.stringify(input.languages) : null, input.timezone ?? null, input.locationCity ?? null, input.locationCountry ?? null, input.headshotUrl ?? null, input.sourceUrl ?? null, isOperator ? 1 : 0, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.identity", + sourceUrl: input.sourceUrl ?? "operator://asserted", + valueJson: { + full_name: input.fullName ?? null, + preferred_name: input.preferredName ?? null, + pronouns: input.pronouns ?? null, + birth_year: input.birthYear ?? null, + nationality: input.nationality ?? null, + languages: input.languages ?? null, + timezone: input.timezone ?? null, + location_city: input.locationCity ?? null, + location_country: input.locationCountry ?? null, + headshot_url: input.headshotUrl ?? null, + is_operator_asserted: isOperator, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 2. addCareerEntry — natural key (entity_id, organization_*, started_at). +// ========================================================================= +export async function addCareerEntry(env, input) { + requireNonEmpty("addCareerEntry", "entityId", input.entityId); + requireNonEmpty("addCareerEntry", "organizationName", input.organizationName); + const sourceUrl = requireSourceUrl("addCareerEntry", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO career_history ( + id, entity_id, organization_entity_id, organization_name, role_title, + seniority, department, started_at, ended_at, is_current, summary, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, COALESCE(organization_entity_id,''), COALESCE(started_at,'')) DO UPDATE SET + organization_name = COALESCE(excluded.organization_name, career_history.organization_name), + role_title = COALESCE(excluded.role_title, career_history.role_title), + seniority = COALESCE(excluded.seniority, career_history.seniority), + department = COALESCE(excluded.department, career_history.department), + ended_at = COALESCE(excluded.ended_at, career_history.ended_at), + is_current = excluded.is_current, + summary = COALESCE(excluded.summary, career_history.summary), + confidence = MAX(excluded.confidence, career_history.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.roleTitle ?? null, input.seniority ?? null, input.department ?? null, input.startedAt ?? null, input.endedAt ?? null, input.isCurrent ? 1 : 0, input.summary ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.career", + sourceUrl, + valueJson: { + organization_entity_id: input.organizationEntityId ?? null, + organization_name: input.organizationName, + role_title: input.roleTitle ?? null, + seniority: input.seniority ?? null, + department: input.department ?? null, + started_at: input.startedAt ?? null, + ended_at: input.endedAt ?? null, + is_current: input.isCurrent === true, + }, + confidence: input.confidence, + observedAt: now, + }); + }); + // Task #8: career changes materially affect persona match scores + // (title, seniority, function, employer industry/size/stage). Fire + // the debounced re-match trigger. + try { + const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); + void triggerEntityMatchRefresh(env, input.entityId).catch(() => undefined); + } + catch { /* trigger is best-effort */ } +} +// ========================================================================= +// 3. addBoardSeat — natural key (entity_id, organization_name, started_at). +// ========================================================================= +export async function addBoardSeat(env, input) { + requireNonEmpty("addBoardSeat", "entityId", input.entityId); + requireNonEmpty("addBoardSeat", "organizationName", input.organizationName); + const sourceUrl = requireSourceUrl("addBoardSeat", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO board_seats ( + id, entity_id, organization_entity_id, organization_name, role, + is_independent, committee, started_at, ended_at, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, organization_name, COALESCE(started_at,'')) DO UPDATE SET + organization_entity_id = COALESCE(excluded.organization_entity_id, board_seats.organization_entity_id), + role = COALESCE(excluded.role, board_seats.role), + is_independent = excluded.is_independent, + committee = COALESCE(excluded.committee, board_seats.committee), + ended_at = COALESCE(excluded.ended_at, board_seats.ended_at), + confidence = MAX(excluded.confidence, board_seats.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.role ?? null, input.isIndependent ? 1 : 0, input.committee ?? null, input.startedAt ?? null, input.endedAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.board_seat", + sourceUrl, + valueJson: { + organization_entity_id: input.organizationEntityId ?? null, + organization_name: input.organizationName, + role: input.role ?? null, + is_independent: input.isIndependent === true, + committee: input.committee ?? null, + started_at: input.startedAt ?? null, + ended_at: input.endedAt ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 4. addEducation — natural key (entity_id, institution, degree, ended_year). +// ========================================================================= +export async function addEducation(env, input) { + requireNonEmpty("addEducation", "entityId", input.entityId); + requireNonEmpty("addEducation", "institution", input.institution); + const sourceUrl = requireSourceUrl("addEducation", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO education_history ( + id, entity_id, institution, degree, field, started_year, ended_year, + honors, source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, institution, COALESCE(degree,''), COALESCE(ended_year,0)) DO UPDATE SET + field = COALESCE(excluded.field, education_history.field), + started_year = COALESCE(excluded.started_year, education_history.started_year), + honors = COALESCE(excluded.honors, education_history.honors), + confidence = MAX(excluded.confidence, education_history.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.institution, input.degree ?? null, input.field ?? null, input.startedYear ?? null, input.endedYear ?? null, input.honors ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.education", + sourceUrl, + valueJson: { + institution: input.institution, + degree: input.degree ?? null, + field: input.field ?? null, + started_year: input.startedYear ?? null, + ended_year: input.endedYear ?? null, + honors: input.honors ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 5. addFamilyTie — natural key (entity_id, relation_type, related_name). +// +// Privacy gate (task contract, "No PII leakage"): +// * `isPublic` is required — no defaulting (the helper throws if it's +// undefined / not a boolean), so a caller can never implicitly create +// a private row by omission. +// * `isPublic === false` additionally requires `isOperatorAsserted === +// true`. This is the explicit operator-intent marker the task calls +// out: only the human operator (via an authenticated route handler +// that sets this flag) can stash a private relationship. Background +// enrichment, agents, and scrapers will never have it set and will +// be rejected. +// * Private rows are NEVER mirrored into `facts` — `facts` is the +// public/agent retrieval surface, and routing PII through it would +// undermine row-level filtering downstream. The structured +// `family_ties` row is the only store for private ties, and the +// route layer is responsible for gating reads by operator identity. +// ========================================================================= +export async function addFamilyTie(env, input) { + requireNonEmpty("addFamilyTie", "entityId", input.entityId); + requireNonEmpty("addFamilyTie", "relationType", input.relationType); + requireNonEmpty("addFamilyTie", "relatedName", input.relatedName); + if (typeof input.isPublic !== "boolean") { + throw new Error("profile.addFamilyTie: isPublic is required and must be an explicit boolean"); + } + if (input.isPublic === false && input.isOperatorAsserted !== true) { + throw new Error("profile.addFamilyTie: private family ties (isPublic=false) require " + + "isOperatorAsserted=true; background enrichers and agents cannot store private relationships"); + } + const sourceUrl = requireSourceUrl("addFamilyTie", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO family_ties ( + id, entity_id, relation_type, related_name, related_entity_id, notes, + is_public, source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, relation_type, related_name) DO UPDATE SET + related_entity_id = COALESCE(excluded.related_entity_id, family_ties.related_entity_id), + notes = COALESCE(excluded.notes, family_ties.notes), + is_public = excluded.is_public, + confidence = MAX(excluded.confidence, family_ties.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.relationType, input.relatedName, input.relatedEntityId ?? null, input.notes ?? null, input.isPublic ? 1 : 0, sourceUrl, input.confidence ?? 1.0, now, now).run(); + // PII firewall: only public ties are projected to `facts`. The + // structured row above is still written for the operator-only UI. + if (input.isPublic === true) { + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.family_tie", + sourceUrl, + valueJson: { + relation_type: input.relationType, + related_name: input.relatedName, + related_entity_id: input.relatedEntityId ?? null, + notes: input.notes ?? null, + is_public: true, + }, + confidence: input.confidence, + observedAt: now, + }); + } + }); +} +// ========================================================================= +// 6. addPreference — upsert on (entity_id, preference_key). +// Mirrors to person.preference.{preferenceKey}; that dynamic predicate +// MUST exist in the registry (validated in mirrorFact + smoke test). +// ========================================================================= +export async function addPreference(env, input) { + requireNonEmpty("addPreference", "entityId", input.entityId); + requireNonEmpty("addPreference", "preferenceKey", input.preferenceKey); + const sourceUrl = requireSourceUrl("addPreference", input.sourceUrl); + const predicate = `person.preference.${input.preferenceKey}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_preferences ( + id, entity_id, preference_key, value_text, value_json, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, preference_key) DO UPDATE SET + value_text = COALESCE(excluded.value_text, person_preferences.value_text), + value_json = COALESCE(excluded.value_json, person_preferences.value_json), + source_url = excluded.source_url, + confidence = MAX(excluded.confidence, person_preferences.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.preferenceKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.valueText ?? null, + valueJson: input.valueJson ?? null, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 7. addInterest — natural key (entity_id, interest_category, interest_value). +// ========================================================================= +export async function addInterest(env, input) { + requireNonEmpty("addInterest", "entityId", input.entityId); + requireNonEmpty("addInterest", "interestCategory", input.interestCategory); + requireNonEmpty("addInterest", "interestValue", input.interestValue); + const sourceUrl = requireSourceUrl("addInterest", input.sourceUrl); + const predicate = `person.interest.${input.interestCategory}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_interests ( + id, entity_id, interest_category, interest_value, weight, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, interest_category, interest_value) DO UPDATE SET + weight = MAX(excluded.weight, person_interests.weight), + confidence = MAX(excluded.confidence, person_interests.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.interestCategory, input.interestValue, input.weight ?? 1.0, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.interestValue, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 8. addLifestyleSignal — natural key (entity_id, signal_key, observed_at). +// ========================================================================= +export async function addLifestyleSignal(env, input) { + requireNonEmpty("addLifestyleSignal", "entityId", input.entityId); + requireNonEmpty("addLifestyleSignal", "signalKey", input.signalKey); + const sourceUrl = requireSourceUrl("addLifestyleSignal", input.sourceUrl); + const predicate = `person.lifestyle.${input.signalKey}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO lifestyle_signals ( + id, entity_id, signal_key, value_text, value_json, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, signal_key) DO UPDATE SET + value_text = COALESCE(excluded.value_text, lifestyle_signals.value_text), + value_json = COALESCE(excluded.value_json, lifestyle_signals.value_json), + source_url = excluded.source_url, + confidence = MAX(excluded.confidence, lifestyle_signals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.valueText ?? null, + valueJson: input.valueJson ?? null, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 9. addTravelPattern — natural key (entity_id, pattern_kind, place, starts_at). +// ========================================================================= +export async function addTravelPattern(env, input) { + requireNonEmpty("addTravelPattern", "entityId", input.entityId); + requireNonEmpty("addTravelPattern", "patternKind", input.patternKind); + requireNonEmpty("addTravelPattern", "place", input.place); + const sourceUrl = requireSourceUrl("addTravelPattern", input.sourceUrl); + const predicate = `person.travel.${input.patternKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO travel_patterns ( + id, entity_id, pattern_kind, place, country_iso2, starts_at, ends_at, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, pattern_kind, place, COALESCE(starts_at,'')) DO UPDATE SET + country_iso2 = COALESCE(excluded.country_iso2, travel_patterns.country_iso2), + ends_at = COALESCE(excluded.ends_at, travel_patterns.ends_at), + confidence = MAX(excluded.confidence, travel_patterns.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.patternKind, input.place, input.countryIso2 ?? null, input.startsAt ?? null, input.endsAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.place, + valueJson: { + country_iso2: input.countryIso2 ?? null, + starts_at: input.startsAt ?? null, + ends_at: input.endsAt ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 10. addConferenceAttendance — UNIQUE(entity_id, conference_name, year). +// ========================================================================= +export async function addConferenceAttendance(env, input) { + requireNonEmpty("addConferenceAttendance", "entityId", input.entityId); + requireNonEmpty("addConferenceAttendance", "conferenceName", input.conferenceName); + if (!Number.isInteger(input.year) || input.year < 1900 || input.year > 2100) { + throw new Error("profile.addConferenceAttendance: year must be a 4-digit integer"); + } + const sourceUrl = requireSourceUrl("addConferenceAttendance", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO conference_attendance ( + id, entity_id, conference_name, year, role, session_topic, city, country_iso2, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, conference_name, year) DO UPDATE SET + role = COALESCE(excluded.role, conference_attendance.role), + session_topic = COALESCE(excluded.session_topic, conference_attendance.session_topic), + city = COALESCE(excluded.city, conference_attendance.city), + country_iso2 = COALESCE(excluded.country_iso2, conference_attendance.country_iso2), + confidence = MAX(excluded.confidence, conference_attendance.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.conferenceName, input.year, input.role ?? null, input.sessionTopic ?? null, input.city ?? null, input.countryIso2 ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.conference", + sourceUrl, + valueJson: { + conference_name: input.conferenceName, + year: input.year, + role: input.role ?? null, + session_topic: input.sessionTopic ?? null, + city: input.city ?? null, + country_iso2: input.countryIso2 ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 11. addGoal — natural key (entity_id, goal_kind, goal_text). +// ========================================================================= +export async function addGoal(env, input) { + requireNonEmpty("addGoal", "entityId", input.entityId); + requireNonEmpty("addGoal", "goalKind", input.goalKind); + requireNonEmpty("addGoal", "goalText", input.goalText); + const sourceUrl = requireSourceUrl("addGoal", input.sourceUrl); + const predicate = `person.goal.${input.goalKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_goals ( + id, entity_id, goal_kind, goal_text, target_date, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, goal_kind, goal_text) DO UPDATE SET + target_date = COALESCE(excluded.target_date, person_goals.target_date), + confidence = MAX(excluded.confidence, person_goals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.goalKind, input.goalText, input.targetDate ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.goalText, + valueJson: { target_date: input.targetDate ?? null }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 12. addConversationHook — natural key (entity_id, hook_kind, hook_text). +// ========================================================================= +export async function addConversationHook(env, input) { + requireNonEmpty("addConversationHook", "entityId", input.entityId); + requireNonEmpty("addConversationHook", "hookKind", input.hookKind); + requireNonEmpty("addConversationHook", "hookText", input.hookText); + const sourceUrl = requireSourceUrl("addConversationHook", input.sourceUrl); + const predicate = `person.hook.${input.hookKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO conversation_hooks ( + id, entity_id, hook_kind, hook_text, related_entity_id, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, hook_kind, hook_text) DO UPDATE SET + related_entity_id = COALESCE(excluded.related_entity_id, conversation_hooks.related_entity_id), + confidence = MAX(excluded.confidence, conversation_hooks.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.hookKind, input.hookText, input.relatedEntityId ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.hookText, + valueJson: { related_entity_id: input.relatedEntityId ?? null }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 13. addAppreciationSignal — natural key (entity_id, signal_kind, signal_text). +// ========================================================================= +export async function addAppreciationSignal(env, input) { + requireNonEmpty("addAppreciationSignal", "entityId", input.entityId); + requireNonEmpty("addAppreciationSignal", "signalKind", input.signalKind); + requireNonEmpty("addAppreciationSignal", "signalText", input.signalText); + const sourceUrl = requireSourceUrl("addAppreciationSignal", input.sourceUrl); + const predicate = `person.appreciation.${input.signalKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO appreciation_signals ( + id, entity_id, signal_kind, signal_text, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, signal_kind, signal_text) DO UPDATE SET + confidence = MAX(excluded.confidence, appreciation_signals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKind, input.signalText, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.signalText, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// EntityService facade — single import surface for callers. +export const EntityService = { + setPersonIdentity, + addCareerEntry, + addBoardSeat, + addEducation, + addFamilyTie, + addPreference, + addInterest, + addLifestyleSignal, + addTravelPattern, + addConferenceAttendance, + addGoal, + addConversationHook, + addAppreciationSignal, +}; +export { EMITTED_PREDICATES }; diff --git a/apps/worker/test-dist-q/entities/query.js b/apps/worker/test-dist-q/entities/query.js new file mode 100644 index 00000000..79f93c4f --- /dev/null +++ b/apps/worker/test-dist-q/entities/query.js @@ -0,0 +1,154 @@ +// High-level read helpers. `loadEntity` returns the canonical envelope +// every consumer can rely on; `searchEntities` is a thin filter DSL that +// joins entity_summary + entity_tags for sub-50ms list responses. +import { getEffectiveFacts, loadCurrentOverrides } from "./facts"; +export async function loadEntity(env, id, opts) { + const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first(); + if (!ent) + return null; + // Task #3 (Editable Profiles): use the SAME shared resolver as the + // summary rebuilder so the two read sites cannot drift. The resolver + // returns one EffectiveFact list with canonical rows + dethroned + // attempts marked overridden_attempt=true; we split that into + // facts[] (canonical) and attempts[] (diff strip). + const [roles, effective, channels, tags, summary, overridesMap] = await Promise.all([ + env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(id).all(), + getEffectiveFacts(env, id, { includeNonCurrent: !!opts?.includeNonCurrent, limit: 500 }), + env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified, is_dnc FROM channels WHERE entity_id = ?`).bind(id).all(), + env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(id).all(), + env.DB.prepare(`SELECT * FROM entity_summary WHERE entity_id = ?`).bind(id).first(), + loadCurrentOverrides(env, id), + ]); + const factRows = []; + const attempts = []; + for (const e of effective) { + const row = { + id: e.id, predicate: e.predicate, + value_text: e.value_text, value_number: e.value_number, + value_json: e.value_json, value_entity_id: e.value_entity_id, + source: e.source, source_kind: e.source_kind, + confidence: e.confidence, verified_score: e.verified_score, + observed_at: e.observed_at, is_current: e.is_current, + superseded_by_override: e.superseded_by_override, + }; + if (e.overridden_attempt) + attempts.push(row); + else + factRows.push(row); + } + const overrideArr = Array.from(overridesMap.values()).map((ov) => ({ + id: ov.id, + predicate: ov.predicate, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: ov.value_json ? (() => { try { + return JSON.parse(ov.value_json); + } + catch { + return ov.value_json; + } })() : null, + overridden_at: ov.overridden_at, + })); + return { + id, kind: ent.kind, entity: ent, + roles: (roles.results ?? []), + facts: factRows, + attempts, + channels: (channels.results ?? []), + tags: (tags.results ?? []), + summary: summary ?? null, + overrides: overrideArr, + }; +} +export async function searchEntities(env, f) { + // IMPORTANT: bind order must match SQL placeholder order. Since the + // tag JOINs appear *before* the WHERE clause in the final SQL, their + // binds must be pushed first. We build two separate bind arrays and + // concatenate them in SQL order at the end. + const joinBinds = []; + const whereBinds = []; + const where = ["s.status = 'active'"]; + if (f.kind) { + where.push("s.kind = ?"); + whereBinds.push(f.kind); + } + if (f.role) { + where.push("s.primary_role = ?"); + whereBinds.push(f.role); + } + if (f.country_iso2) { + where.push("s.country_iso2 = ?"); + whereBinds.push(f.country_iso2.toUpperCase()); + } + if (typeof f.check_min_usd === "number") { + where.push("s.check_size_max_usd >= ?"); + whereBinds.push(f.check_min_usd); + } + if (typeof f.check_max_usd === "number") { + where.push("s.check_size_min_usd <= ?"); + whereBinds.push(f.check_max_usd); + } + if (f.has_unicorn) { + where.push("s.unicorn_count > 0"); + } + if (typeof f.min_fit === "number") { + where.push("s.fit_max_score >= ?"); + whereBinds.push(f.min_fit); + } + if (typeof f.min_intent === "number") { + where.push("s.intent_score >= ?"); + whereBinds.push(f.min_intent); + } + if (f.q) { + where.push("(lower(s.display_name) LIKE ? OR lower(s.primary_domain) LIKE ? OR lower(s.primary_email) LIKE ?)"); + const q = `%${f.q.toLowerCase()}%`; + whereBinds.push(q, q, q); + } + // Tag filters require a JOIN per taxonomy so we can intersect. + // NOTE: `has_role` is *not* a tag — roles live in entity_roles + // (addRole writes there, not entity_tags). The previous JOIN on + // entity_tags taxonomy='role' returned empty results for every + // wrapper helper (listFirms, listInvestors, listCompanies, + // listAccounts, listBuyers, listFounders). We now JOIN entity_roles + // directly for role membership. + const joins = []; + let tagJoinIdx = 0; + for (const [tax, slug] of [["sector", f.sector], ["stage", f.stage], ["geo", f.geo]]) { + if (!slug) + continue; + const alias = `t${++tagJoinIdx}`; + joins.push(`JOIN entity_tags ${alias} ON ${alias}.entity_id = s.entity_id AND ${alias}.taxonomy = ? AND ${alias}.slug = ?`); + joinBinds.push(tax, slug); + } + if (f.has_role) { + joins.push(`JOIN entity_roles er ON er.entity_id = s.entity_id AND er.role = ?`); + joinBinds.push(f.has_role); + } + const sortCol = (() => { + switch (f.sort) { + case "intent": return "s.intent_score DESC"; + case "quality": return "s.quality_score DESC"; + case "updated": return "s.rebuilt_at DESC"; + default: return "s.fit_max_score DESC"; + } + })(); + const limit = Math.min(Math.max(1, f.limit ?? 50), 200); + const offset = Math.max(0, f.offset ?? 0); + const sql = `SELECT s.* FROM entity_summary s ${joins.join(" ")} + WHERE ${where.join(" AND ")} + ORDER BY ${sortCol}, s.entity_id ASC + LIMIT ? OFFSET ?`; + const r = await env.DB.prepare(sql).bind(...joinBinds, ...whereBinds, limit + 1, offset).all(); + const rows = r.results ?? []; + const hasMore = rows.length > limit; + return { + items: (hasMore ? rows.slice(0, limit) : rows), + next_offset: hasMore ? offset + limit : null, + }; +} +export const listFirms = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "firm" }); +export const listInvestors = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "investor" }); +export const listCompanies = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "company" }); +export const listAccounts = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: "account" }); +export const listBuyers = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "buyer" }); +export const listFounders = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "founder" }); diff --git a/apps/worker/test-dist-q/entities/roles.js b/apps/worker/test-dist-q/entities/roles.js new file mode 100644 index 00000000..2b16577f --- /dev/null +++ b/apps/worker/test-dist-q/entities/roles.js @@ -0,0 +1,120 @@ +import { isGarbage, logDataQuality, classifyPersonName } from "./garbage"; +export async function createEntity(env, init) { + // Task #9: pre-insert garbage guard. The pure heuristic detector + // (no AI call here — keep createEntity synchronous and cheap on the + // hot write path) rejects HTML page titles / nav strings / UI + // labels that the crawler may have mistaken for entity names. + // Returns null + audit row instead of throwing so callers can + // skip without crashing the broader import. The AI second opinion + // runs only in the cron sweep, NOT inline on every write. + const verdict = isGarbage({ + kind: init.kind, + display_name: init.display_name ?? null, + primary_url: init.primary_url ?? null, + primary_domain: init.primary_domain ?? null, + primary_email_key: init.primary_email_key ?? null, + primary_linkedin_key: init.primary_linkedin_key ?? null, + }); + if (verdict.is_garbage) { + console.log("garbage.pre_insert_rejected", JSON.stringify({ + kind: init.kind, display_name: init.display_name, reasons: verdict.reasons, + })); + // Log to data_quality_log with a synthetic entity_id so operators + // can audit rejected writes too. We use a `rejected:` prefix to + // distinguish from soft-deleted entities (which carry real ids). + void logDataQuality(env, "rejected:" + (init.display_name ?? "").slice(0, 100), "pre_insert_rejected", verdict.reasons, "pre_insert_guard", null).catch(() => undefined); + return null; + } + // Task #6: reclassify-on-write. A `person` whose display name is clearly + // an organization ("Intel Capital", "Mendoza Ventures") is written as an + // `org` so it never lands in the People list in the first place. A strong + // personal identifier (personal LinkedIn /in/ or email) contradicting the + // org-suffix name suppresses the flip — never mislabel a likely real + // person. Junk names were already rejected by the isGarbage guard above. + let effectiveKind = init.kind; + let reclassifiedOrgRole = null; + if (init.kind === "person") { + const cls = classifyPersonName(init.display_name ?? null); + if (cls.verdict === "organization" && cls.orgRole) { + const personalLinkedin = !!init.primary_linkedin_key && /(^|\/)in\//i.test(init.primary_linkedin_key); + const hasEmail = !!init.primary_email_key; + if (!personalLinkedin && !hasEmail) { + effectiveKind = "org"; + reclassifiedOrgRole = cls.orgRole; + } + } + } + const id = crypto.randomUUID(); + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO u_entities ( + id, kind, display_name, primary_url, primary_domain, + primary_email_key, primary_linkedin_key, primary_twitter_handle, primary_github_handle, + status, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'active', ?, ?)`).bind(id, effectiveKind, init.display_name ?? null, init.primary_url ?? null, init.primary_domain ?? null, init.primary_email_key ?? null, init.primary_linkedin_key ?? null, init.primary_twitter_handle ?? null, init.primary_github_handle ?? null, now, now).run(); + await env.DB.prepare(`INSERT INTO entity_history (id, entity_id, action, source, changed_at) VALUES (?, ?, 'create', 'system', ?)`).bind(crypto.randomUUID(), id, now).run(); + // Task #8: enqueue persona ↔ entity matching for newly-created + // person entities so a freshly-created founder/operator appears in + // matching personas' candidate lists within minutes — even before + // any career/title facts are written. KV-debounced inside trigger. + if (effectiveKind === "person") { + try { + const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); + void triggerEntityMatchRefresh(env, id).catch(() => undefined); + } + catch { /* best-effort */ } + } + // Task #6: stamp the inferred org role + an audit row when a person was + // reclassified to an org on write, so the operator console can trace it. + if (reclassifiedOrgRole) { + await addRole(env, id, reclassifiedOrgRole, { is_primary: true, source: "garbage_reclassify_on_write" }); + void logDataQuality(env, id, "reclassified", [`org_role:${reclassifiedOrgRole}`, "reclassified_on_write"], "pre_insert_guard", null).catch(() => undefined); + } + // Task #3 (AI Profile Filler): auto-trigger a profile fill for newly + // created org entities that have a website but no facts yet (the + // signal-poor "low confidence" case the spec calls out). Dispatched + // via WF binding when available so the cost is async and respects + // the daily neuron cap. No-op when the binding isn't configured. + if (effectiveKind === "org" && (init.primary_url || init.primary_domain) && !init.suppressAutoProfileFill) { + const wf = env.WF_PROFILE_FILLER; + if (wf) { + try { + void wf.create({ params: { entityId: id, force: false, triggeredBy: "auto:entity_created" } }).catch(() => undefined); + } + catch { /* best-effort */ } + } + } + // Task #4 (Relationship Inference Worker): debounced enqueue into + // relationship_infer_queue (migration 377). The consolidated nightly + // slot drains the queue with the per-entity orchestrator pass. + try { + const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); + void enqueueRelInfer(env, id, `created:${effectiveKind}`).catch(() => undefined); + } + catch { /* best-effort */ } + return (await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first()); +} +export async function addRole(env, entityId, role, opts) { + try { + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, role) DO UPDATE SET + is_primary = MAX(is_primary, excluded.is_primary), + confidence = MAX(confidence, excluded.confidence)`).bind(entityId, role, opts?.is_primary ? 1 : 0, opts?.source ?? null, opts?.confidence ?? 1).run(); + } + catch (e) { + console.warn("addRole failed", role, e.message); + } +} +export async function getLegacyEntityId(env, table, legacyId) { + const r = await env.DB.prepare(`SELECT entity_id FROM entity_legacy_map WHERE legacy_table = ? AND legacy_id = ?`).bind(table, String(legacyId)).first(); + return r?.entity_id ?? null; +} +export async function setLegacyEntityId(env, table, legacyId, entityId) { + try { + await env.DB.prepare(`INSERT INTO entity_legacy_map (legacy_table, legacy_id, entity_id) + VALUES (?, ?, ?) ON CONFLICT DO NOTHING`).bind(table, String(legacyId), entityId).run(); + } + catch (e) { + console.warn("setLegacyEntityId failed", table, legacyId, e.message); + } +} diff --git a/apps/worker/test-dist-q/entities/summary.js b/apps/worker/test-dist-q/entities/summary.js new file mode 100644 index 00000000..fc6eef0b --- /dev/null +++ b/apps/worker/test-dist-q/entities/summary.js @@ -0,0 +1,121 @@ +// Rebuild `entity_summary` for one entity from its current facts + +// channels + tags + roles. Runs inside the queue consumer. +import { getEffectiveFacts } from "./facts"; +const SOURCE_PRIORITY = { + manual: 5, enrichment: 4, import: 3, scrape: 2, ai: 1, inferred: 0, +}; +function pickBestFact(rows) { + if (!rows.length) + return null; + return rows.slice().sort((a, b) => { + const sa = (a.confidence ?? 0) * 100 + (SOURCE_PRIORITY[a.source_kind] ?? 0) * 10 + Date.parse(a.observed_at) / 1e12; + const sb = (b.confidence ?? 0) * 100 + (SOURCE_PRIORITY[b.source_kind] ?? 0) * 10 + Date.parse(b.observed_at) / 1e12; + return sb - sa; + })[0]; +} +function txt(rows, predicate) { + const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); + return f?.value_text ?? null; +} +function num(rows, predicate) { + const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); + return f?.value_number ?? null; +} +export async function rebuildSummary(env, entityId) { + const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(entityId).first(); + if (!ent) + return false; + if (ent.status === "merged" || ent.status === "soft_deleted") { + await env.DB.prepare(`DELETE FROM entity_summary WHERE entity_id = ?`).bind(entityId).run(); + return true; + } + // Task #3 (Editable Profiles): the summary input is the EFFECTIVE + // facts view — overrides win, overridden_attempt rows are filtered. + // Same resolver as the per-entity read path in query.ts, so the two + // call sites cannot drift. + const [effective, tagsRes, rolesRes, channelsRes] = await Promise.all([ + getEffectiveFacts(env, entityId), + env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(entityId).all(), + env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(), + env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified FROM channels WHERE entity_id = ?`).bind(entityId).all(), + ]); + const facts = effective + .filter((e) => !e.overridden_attempt) + .map((e) => ({ + predicate: e.predicate, + value_text: e.value_text, + value_number: e.value_number, + value_json: e.value_json != null ? (typeof e.value_json === "string" ? e.value_json : JSON.stringify(e.value_json)) : null, + value_entity_id: e.value_entity_id, + confidence: e.confidence, + observed_at: e.observed_at, + source_kind: e.source_kind, + })); + const tags = tagsRes.results ?? []; + const roles = rolesRes.results ?? []; + const channels = channelsRes.results ?? []; + const primaryRole = (roles.find((r) => r.is_primary === 1) ?? roles[0])?.role ?? null; + const display = ent.display_name ?? txt(facts, "name") ?? txt(facts, "display_name"); + const country = txt(facts, "country_iso2"); + const region = txt(facts, "region"); + const city = txt(facts, "city"); + const sectors = tags.filter((t) => t.taxonomy === "sector").map((t) => t.slug); + const stages = tags.filter((t) => t.taxonomy === "stage").map((t) => t.slug); + const geos = tags.filter((t) => t.taxonomy === "geo").map((t) => t.slug); + const checkMin = num(facts, "check_size_min_usd"); + const checkMax = num(facts, "check_size_max_usd"); + const fitMax = num(facts, "fit_max_score") ?? 0; + const intent = num(facts, "intent_score") ?? 0; + const unicornCount = Math.round(num(facts, "unicorn_count") ?? 0); + const primaryEmailChan = channels.filter((c) => c.kind === "email").sort((a, b) => (b.is_primary - a.is_primary) || (b.is_verified - a.is_verified))[0]; + const primaryLinkedinChan = channels.filter((c) => c.kind === "linkedin")[0]; + const employerEntityId = (facts.find((f) => f.predicate === "employer" && f.value_entity_id) ?? null)?.value_entity_id ?? null; + let employerDisplay = null; + if (employerEntityId) { + const r = await env.DB.prepare(`SELECT display_name FROM u_entities WHERE id = ?`).bind(employerEntityId).first(); + employerDisplay = r?.display_name ?? null; + } + else { + employerDisplay = txt(facts, "primary_employer") ?? txt(facts, "org"); + } + // Quality: coverage × confidence average × source diversity, scaled 0..100. + const coverage = Math.min(1, facts.length / 12); + const confAvg = facts.length ? facts.reduce((s, f) => s + (f.confidence ?? 0), 0) / facts.length : 0; + const diversity = Math.min(1, new Set(facts.map((f) => f.source_kind)).size / 3); + const quality = Math.round((coverage * 0.5 + confAvg * 0.3 + diversity * 0.2) * 100); + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO entity_summary ( + entity_id, kind, display_name, primary_role, primary_employer, primary_employer_entity_id, + country_iso2, region, city, sectors_csv, stages_csv, geos_csv, + check_size_min_usd, check_size_max_usd, primary_email, primary_linkedin, + primary_domain, quality_score, fit_max_score, intent_score, unicorn_count, status, rebuilt_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id) DO UPDATE SET + kind = excluded.kind, + display_name = excluded.display_name, + primary_role = excluded.primary_role, + primary_employer = excluded.primary_employer, + primary_employer_entity_id = excluded.primary_employer_entity_id, + country_iso2 = excluded.country_iso2, + region = excluded.region, + city = excluded.city, + sectors_csv = excluded.sectors_csv, + stages_csv = excluded.stages_csv, + geos_csv = excluded.geos_csv, + check_size_min_usd = excluded.check_size_min_usd, + check_size_max_usd = excluded.check_size_max_usd, + primary_email = excluded.primary_email, + primary_linkedin = excluded.primary_linkedin, + primary_domain = excluded.primary_domain, + quality_score = excluded.quality_score, + fit_max_score = excluded.fit_max_score, + intent_score = excluded.intent_score, + unicorn_count = excluded.unicorn_count, + status = excluded.status, + rebuilt_at = excluded.rebuilt_at`).bind(entityId, ent.kind, display, primaryRole, employerDisplay, employerEntityId, country, region, city, sectors.join(","), stages.join(","), geos.join(","), checkMin != null ? Math.round(checkMin) : null, checkMax != null ? Math.round(checkMax) : null, primaryEmailChan?.canonical ?? null, primaryLinkedinChan?.canonical ?? null, ent.primary_domain, quality, fitMax, intent, unicornCount, ent.status, now).run(); + // Update u_entities.quality_score + last_summary_at so list paths that + // still read from `u_entities` see the latest score. + await env.DB.prepare(`UPDATE u_entities SET quality_score = ?, last_summary_at = ?, updated_at = ? WHERE id = ?`) + .bind(quality, now, now, entityId).run(); + return true; +} diff --git a/apps/worker/test-dist-q/entities/summaryQueue.js b/apps/worker/test-dist-q/entities/summaryQueue.js new file mode 100644 index 00000000..ccbf3c4b --- /dev/null +++ b/apps/worker/test-dist-q/entities/summaryQueue.js @@ -0,0 +1,26 @@ +// Enqueue + consume `rebuild_summary` work. Uses the existing LEAD_QUEUE +// to avoid provisioning a second queue; the consumer in index.ts +// dispatches by message shape. +import { rebuildSummary } from "./summary"; +export function isRebuildSummaryMessage(m) { + return !!m && typeof m === "object" && m.type === "rebuild_summary" + && typeof m.entityId === "string"; +} +// We intentionally do *not* debounce via KV: the previous design dropped +// later writes when a debounce key was already set, which left +// entity_summary stale until the next mutation. The queue handler can +// coalesce duplicates at consume time if needed; in the meantime, the +// extra work is one upsert into a tiny rollup table. +export async function enqueueSummaryRebuild(env, entityId) { + if (!entityId) + return; + try { + await env.LEAD_QUEUE.send({ type: "rebuild_summary", entityId }); + } + catch (e) { + console.warn("enqueueSummaryRebuild failed", entityId, e.message); + } +} +export async function handleSummaryMessage(env, m) { + await rebuildSummary(env, m.entityId); +} diff --git a/apps/worker/test-dist-q/entities/tags.js b/apps/worker/test-dist-q/entities/tags.js new file mode 100644 index 00000000..a28e6fe3 --- /dev/null +++ b/apps/worker/test-dist-q/entities/tags.js @@ -0,0 +1,37 @@ +export async function addTag(env, t) { + if (!t.entity_id || !t.slug) + return; + const slug = String(t.slug).trim().toLowerCase(); + if (!slug) + return; + try { + await env.DB.prepare(`INSERT INTO entity_tags (entity_id, taxonomy, slug, weight, source) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, taxonomy, slug) + DO UPDATE SET weight = MAX(weight, excluded.weight)`).bind(t.entity_id, t.taxonomy, slug, t.weight ?? 1, t.source ?? null).run(); + } + catch (e) { + console.warn("addTag failed", t.taxonomy, slug, e.message); + } +} +export async function addTagsFromJsonArray(env, entityId, taxonomy, rawJsonArray, source) { + if (!rawJsonArray) + return 0; + let arr; + try { + arr = JSON.parse(rawJsonArray); + } + catch { + return 0; + } + if (!Array.isArray(arr)) + return 0; + let n = 0; + for (const v of arr) { + if (typeof v !== "string") + continue; + await addTag(env, { entity_id: entityId, taxonomy, slug: v, source }); + n += 1; + } + return n; +} diff --git a/apps/worker/test-dist-q/errors.js b/apps/worker/test-dist-q/errors.js new file mode 100644 index 00000000..653c1b1a --- /dev/null +++ b/apps/worker/test-dist-q/errors.js @@ -0,0 +1,328 @@ +// Centralized error taxonomy for the worker (Task #27). +// +// Every operational failure should be either: +// 1. an `AppError` subclass thrown explicitly, OR +// 2. caught and re-thrown via `wrapUnknown(e, code, ctx)` so the global +// onError handler in `index.ts` can serialize it to JSON, log it to +// `error_log`, mirror it to Analytics Engine, and attach a request_id. +// +// All AppErrors carry: +// - code: stable machine string from the ErrCode union below. +// - status: HTTP status to return when surfaced via API. +// - kind: high-level category for UI grouping. +// - retryable: hint to the queue/retry layer. +// - context: free-form structured context (job_id, url, host, provider…). +export class AppError extends Error { + code; + kind; + status; + retryable; + context; + cause; + constructor(opts) { + super(opts.message ?? opts.code); + this.name = "AppError"; + this.code = opts.code; + this.kind = opts.kind; + this.status = opts.status ?? defaultStatusForKind(opts.kind); + this.retryable = opts.retryable ?? defaultRetryableForKind(opts.kind); + this.context = opts.context ?? {}; + if (opts.cause instanceof Error) + this.cause = opts.cause; + } + toJSON(requestId) { + const out = { + error: this.code, + code: this.code, + kind: this.kind, + status: this.status, + message: this.message, + retryable: this.retryable, + }; + if (Object.keys(this.context).length) + out.context = this.context; + if (requestId) + out.request_id = requestId; + if (this.cause) { + const c = { + name: this.cause.name, + message: this.cause.message, + }; + if (this.cause.stack) + c.stack = this.cause.stack; + out.cause = c; + } + return out; + } +} +function defaultStatusForKind(kind) { + switch (kind) { + case "validation": return 400; + case "auth": return 401; + case "permanent": return 422; + case "config": return 500; + case "upstream": return 502; + case "transient": return 503; + // Task #72: a benign skip is not an error surface — neutral 200. + case "skip": return 200; + case "internal": + default: return 500; + } +} +function defaultRetryableForKind(kind) { + return kind === "transient" || kind === "upstream"; +} +// ---- Common subclasses (sugar; AppError directly is also fine) ---------- +export class ValidationError extends AppError { + constructor(code, message, context) { + super({ code, kind: "validation", status: 400, message, retryable: false, ...(context ? { context } : {}) }); + this.name = "ValidationError"; + } +} +export class NotFoundError extends AppError { + constructor(resource, id) { + super({ + code: "not_found", + kind: "permanent", + status: 404, + message: `${resource}${id ? ` ${id}` : ""} not found`, + retryable: false, + context: id ? { resource, id } : { resource }, + }); + this.name = "NotFoundError"; + } +} +export class AuthError extends AppError { + constructor(code, message, context) { + super({ + code, + kind: "auth", + status: code === "forbidden" ? 403 : 401, + message: message ?? code, + retryable: false, + ...(context ? { context } : {}), + }); + this.name = "AuthError"; + } +} +export class UpstreamError extends AppError { + constructor(provider, message, context) { + // Constructed at runtime; the closed ErrCode enum already enumerates + // every known provider so this assertion is the only escape. + const code = `upstream_${provider}`; + super({ + code, + kind: "upstream", + status: 502, + message, + retryable: true, + context: { provider, ...(context ?? {}) }, + }); + this.name = "UpstreamError"; + } +} +export class ScrapeBlockedError extends AppError { + constructor(host, reason, context) { + super({ + code: "scrape_blocked", + kind: "permanent", + status: 403, + message: `${host}: ${reason}`, + retryable: false, + context: { host, reason, ...(context ?? {}) }, + }); + this.name = "ScrapeBlockedError"; + } +} +export class BudgetExhaustedError extends AppError { + constructor(scope, context) { + super({ + code: "budget_exhausted", + kind: "permanent", + status: 429, + message: `Budget exhausted for ${scope}`, + retryable: false, + context: { scope, ...(context ?? {}) }, + }); + this.name = "BudgetExhaustedError"; + } +} +export class TransientError extends AppError { + constructor(code, message, context) { + super({ code, kind: "transient", status: 503, message, retryable: true, ...(context ? { context } : {}) }); + this.name = "TransientError"; + } +} +// ---- Wrappers -------------------------------------------------------------- +/** Convert any unknown thrown value into an AppError (idempotent). */ +export function wrapUnknown(e, code, context) { + if (e instanceof AppError) { + if (context) + Object.assign(e.context, context); + return e; + } + const err = e instanceof Error ? e : new Error(typeof e === "string" ? e : safeJson(e)); + // Heuristic upgrade: detect common transient patterns from the cause. + const guessed = classify(err); + const opts = { + code: guessed?.code ?? code, + kind: guessed?.kind ?? "internal", + message: err.message || code, + cause: err, + }; + if (guessed) + opts.retryable = guessed.retryable; + if (context) + opts.context = context; + return new AppError(opts); +} +/** Type guard. */ +export function isAppError(e) { + return e instanceof AppError; +} +/** + * Heuristic classifier for stringly-typed errors thrown by the existing + * codebase or by the platform (D1, fetch, Workers AI). Returns null if no + * pattern matches; callers should then use the supplied default code/kind. + * + * Required by Task #27 acceptance: every logged failure has a well-typed + * code, even when thrown deep in legacy code that hasn't migrated to + * AppError yet. + */ +export function classify(err) { + if (err instanceof AppError) + return { code: err.code, kind: err.kind, retryable: err.retryable }; + const msg = (err instanceof Error ? err.message : String(err ?? "")).toLowerCase(); + if (!msg) + return null; + // Task #70: Cloudflare's per-invocation subrequest cap surfaces as + // "Too many subrequests by single Worker invocation", which bubbles up + // wrapped in a `fetch_failed:proxy_error:...` (or `fetch_error:...`) + // string. This is NOT a permanent scrape block — the page is fine, the + // invocation just ran out of budget — so it must be classified + // transient/retryable AHEAD of the generic `fetch_failed:` permanent + // rule below. Matched specifically (not all proxy_errors) so genuine + // upstream proxy failures still dead-letter as before. + // + // `subrequest_budget_exhausted` is our OWN pre-emptive refusal (the + // crawl-path budget stopped a fetch before it could trip the platform + // cap); it is the same condition and must retry identically. + if (msg.includes("too many subrequests") || msg.includes("subrequest_budget")) { + return { code: "subrequest_limit", kind: "transient", retryable: true }; + } + // Pipeline-level fetch/scrape sentinels. These reasons are bubbled up + // from the scraper as plain `Error("fetch_failed::status=")` + // (see scraper/pipeline.ts). They are expected operational outcomes — + // not real internal errors — so we map them to typed codes the queue + // can dead-letter without paging. + if (msg.includes("scraping_api_not_configured") || + msg.includes("proxy_not_configured") || + msg.includes("browser_binding_unavailable") || + msg.includes("puppeteer_module_missing")) { + return { code: "config_missing", kind: "config", retryable: false }; + } + // Task #72: robots.txt / ToS blocks are EXPECTED, benign policy outcomes, + // not internal errors — honoring a host's robots.txt is correct behavior. + // Classify them as the `skip` kind so the queue routes them to the `skipped` + // terminal status (no error_log row, never retried) instead of failing / + // dead-lettering them as scrape errors. NB: the scraper emits the token + // `robots_disallow` (see scraper/robots.ts); the old code only matched the + // `robots_disallowed` spelling and silently fell through to the generic + // fetch_failed → permanent rule, so the block surfaced as a red 422. + if (msg.includes("robots_disallow") || msg.includes("tos_blocked")) { + const code = msg.includes("tos_blocked") ? "tos_blocked" : "robots_disallowed"; + return { code, kind: "skip", retryable: false }; + } + // Gated sources still need an operator manual-paste; that's a permanent + // scrape block (the queue preflight already skips it earlier — this is the + // fetcher backstop, kept permanent so its behavior is unchanged). + if (msg.includes("gated_source_use_manual_paste")) { + return { code: "scrape_blocked", kind: "permanent", retryable: false }; + } + if (msg.includes("no_table_found")) { + return { code: "parse_error", kind: "validation", retryable: false }; + } + if (msg.startsWith("fetch_failed:") || msg.includes(":fetch_failed:")) { + // Task #71: a `fetch_failed:` message can carry an embedded upstream HTTP + // status (e.g. "fetch_failed:status_429:status=429"). A 429 rate-limit and + // any 5xx are TRANSIENT — the page is fine, the upstream is just briefly + // refusing — so they must retry with backoff, not get dropped as a + // permanent scrape block on attempt 1. Parse the embedded status BEFORE the + // generic permanent fallback. A genuine 4xx (403/404/...) or a + // fetch_failed with no recoverable status still resolves to permanent + // scrape_blocked (prior behavior preserved). The dedicated scrape sentinels + // (robots/tos/gated/config) matched above stay permanent regardless. + const embedded = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); + if (embedded) { + const s = Number(embedded[1]); + if (s === 429) + return { code: "rate_limited", kind: "transient", retryable: true }; + if (s >= 500) + return { code: "fetch.http_5xx", kind: "transient", retryable: true }; + } + // Generic fetch_failed without a recoverable status → upstream/permanent. + return { code: "scrape_blocked", kind: "permanent", retryable: false }; + } + // Network / fetch. + if (msg.includes("aborted") || msg.includes("timeout") || msg.includes("timed out")) { + return { code: "fetch.timeout", kind: "transient", retryable: true }; + } + if (msg.includes("network connection lost") || msg.includes("econnreset") || msg.includes("ehostunreach")) { + return { code: "fetch.error", kind: "transient", retryable: true }; + } + // HTTP status codes embedded in the message (e.g. "status_503", "status=404", + // "status: 502", or a bare " 404 " token). + const statusMatch = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); + if (statusMatch) { + const s = Number(statusMatch[1]); + if (s === 429) + return { code: "rate_limited", kind: "transient", retryable: true }; + if (s >= 500) + return { code: "fetch.http_5xx", kind: "transient", retryable: true }; + if (s >= 400) + return { code: "fetch.http_4xx", kind: "permanent", retryable: false }; + } + // D1 / Vectorize / Workers AI. + if (msg.includes("d1_error") || msg.includes("sqlite_") || msg.includes("database is locked")) { + return { code: "db_error", kind: "transient", retryable: true }; + } + if (msg.includes("vectorize")) + return { code: "vectorize_error", kind: "upstream", retryable: true }; + if (msg.includes("ai.run") || msg.includes("workers ai")) + return { code: "ai_error", kind: "upstream", retryable: true }; + // Parsing. + if (msg.includes("unexpected token") || msg.includes("json")) + return { code: "json_parse_error", kind: "validation", retryable: false }; + if (msg.includes("invalid url") || msg.includes("uri malformed")) + return { code: "validation_failed", kind: "validation", retryable: false }; + // Auth. + if (msg.includes("jwt") || msg.includes("unauthorized") || msg.includes("forbidden")) { + return { code: "unauthorized", kind: "auth", retryable: false }; + } + return null; +} +/** + * Task #72: is this thrown value a benign policy skip (robots.txt / ToS) + * rather than a real fetch failure? Benign skips end a job in the `skipped` + * terminal status (no error_log, no retry), NOT failed/dead_letter. Returns + * the stable `skip_code` + a human reason, or null for everything else (which + * the caller then classifies / retries / dead-letters normally). + */ +export function isBenignSkip(err) { + const cls = err instanceof AppError + ? { code: err.code, kind: err.kind } + : classify(err); + if (!cls || cls.kind !== "skip") + return null; + const skip_code = cls.code === "tos_blocked" ? "tos_blocked" : "robots_disallow"; + const reason = err instanceof Error ? err.message : String(err ?? skip_code); + return { skip_code, reason }; +} +function safeJson(v) { + try { + return JSON.stringify(v); + } + catch { + return String(v); + } +} diff --git a/apps/worker/test-dist-q/personas/repo.js b/apps/worker/test-dist-q/personas/repo.js new file mode 100644 index 00000000..e0170927 --- /dev/null +++ b/apps/worker/test-dist-q/personas/repo.js @@ -0,0 +1,403 @@ +// Task #46: persona data-access layer. +export const PERSONA_FIELDS = [ + "name", "kind", "status", "thesis", + "hard_filters_json", "size_min", "size_max", "size_bands_json", + "geos_json", "industries_json", + "techs_required_json", "techs_preferred_json", "techs_excluded_json", + "signal_kinds_json", "buyer_titles_json", "buyer_seniority_json", "buyer_departments_json", + "weights_json", "semantic_fit_threshold", "recency_boost", +]; +function parseJsonArr(s) { + if (!s) + return []; + try { + const v = JSON.parse(s); + return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; + } + catch { + return []; + } +} +function parseJsonObj(s) { + if (!s) + return {}; + try { + const v = JSON.parse(s); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +export function rowToSpec(row) { + const w = parseJsonObj(row.weights_json); + // Legacy scorer only understands account/buyer. New taxonomy kinds + // fall through to "buyer" for spec purposes (the kind dispatcher in + // services/personas/kinds owns real matching for new kinds; this + // spec is only used by the legacy persona_matches code path). + const legacyKind = row.kind === "account" || row.kind === "account_company" ? "account" : "buyer"; + return { + id: row.id, + kind: legacyKind, + size_min: row.size_min, + size_max: row.size_max, + size_bands: parseJsonArr(row.size_bands_json), + geos: parseJsonArr(row.geos_json), + industries: parseJsonArr(row.industries_json), + techs_required: parseJsonArr(row.techs_required_json), + techs_preferred: parseJsonArr(row.techs_preferred_json), + techs_excluded: parseJsonArr(row.techs_excluded_json), + signal_kinds: parseJsonArr(row.signal_kinds_json), + buyer_titles: parseJsonArr(row.buyer_titles_json), + buyer_seniority: parseJsonArr(row.buyer_seniority_json), + buyer_departments: parseJsonArr(row.buyer_departments_json), + hard_filters: parseJsonObj(row.hard_filters_json), + weights: w, + semantic_fit_threshold: row.semantic_fit_threshold ?? 0.55, + recency_boost: row.recency_boost ?? 0, + }; +} +export async function listPersonas(env, opts) { + const status = opts?.status ?? "active"; + const limit = Math.min(Math.max(1, opts?.limit ?? 200), 500); + const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = ? ORDER BY last_modified DESC LIMIT ?`).bind(status, limit).all(); + return r.results ?? []; +} +export async function getPersona(env, id) { + const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ? AND deleted_at IS NULL`).bind(id).first(); + return r ?? null; +} +export async function getPersonaIncludingDeleted(env, id) { + const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ?`).bind(id).first(); + return r ?? null; +} +export async function insertPersona(env, body, by, idOverride) { + const id = idOverride ?? crypto.randomUUID(); + const now = new Date().toISOString(); + const cols = ["id", "created_by", "created_at", "updated_at", "last_modified", ...PERSONA_FIELDS]; + const binds = [id, by ?? null, now, now, now]; + // Defaults for NOT NULL columns when the caller omits them (e.g. the + // seed loader). Without this, the seed path threw + // `NOT NULL constraint failed: personas.status` and surfaced as a + // db_error on the Personas page. + const defaults = { status: "active", kind: "account" }; + for (const f of PERSONA_FIELDS) { + const v = body[f]; + binds.push(v ?? defaults[f] ?? null); + } + await env.DB.prepare(`INSERT INTO personas (${cols.join(",")}) VALUES (${cols.map(() => "?").join(",")})`).bind(...binds).run(); + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'created', ?, ?)`) + .bind(crypto.randomUUID(), id, body.name, by ?? null).run(); + const row = await getPersona(env, id); + return row; +} +export async function updatePersona(env, id, patch, by) { + const cur = await getPersona(env, id); + if (!cur) + return null; + const allowed = new Set(PERSONA_FIELDS); + const sets = []; + const binds = []; + const hist = []; + for (const [k, v] of Object.entries(patch)) { + if (!allowed.has(k)) + continue; + sets.push(`${k} = ?`); + binds.push(v); + const before = cur[k]; + if (before !== v) + hist.push({ field: k, old: before, nw: v }); + } + if (!sets.length) + return cur; + const now = new Date().toISOString(); + binds.push(now, now, id); + await env.DB.prepare(`UPDATE personas SET ${sets.join(", ")}, updated_at = ?, last_modified = ? WHERE id = ?`).bind(...binds).run(); + for (const h of hist) { + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, old_value, new_value, changed_by) VALUES (?, ?, ?, ?, ?, ?)`) + .bind(crypto.randomUUID(), id, h.field, h.old != null ? String(h.old) : null, h.nw != null ? String(h.nw) : null, by ?? null).run(); + } + return await getPersona(env, id); +} +export async function setPersonaEmbeddingMeta(env, id, dim, text) { + const now = new Date().toISOString(); + await env.DB.prepare(`UPDATE personas SET embedding_dim = ?, embedded_at = ?, embedding_text = ?, updated_at = ?, last_modified = ? WHERE id = ?`) + .bind(dim, now, text, now, now, id).run(); +} +export async function setPersonaNotes(env, id, notes) { + const now = new Date().toISOString(); + await env.DB.prepare(`UPDATE personas SET persona_notes = ?, notes_generated_at = ?, updated_at = ? WHERE id = ?`) + .bind(notes, now, now, id).run(); +} +export async function softDeletePersona(env, id, by) { + const now = new Date().toISOString(); + const r = await env.DB.prepare(`UPDATE personas SET deleted_at = ?, status = 'archived', updated_at = ? WHERE id = ? AND deleted_at IS NULL`).bind(now, now, id).run(); + if ((r.meta?.changes ?? 0) > 0) { + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'archived', ?, ?)`) + .bind(crypto.randomUUID(), id, now, by ?? null).run(); + return true; + } + return false; +} +export async function upsertMatch(env, args) { + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO persona_matches (persona_id, entity_kind, entity_id, fit_score, hard_filter_pass, components_json, explanation, explanation_at, persona_modified_at, entity_modified_at, computed_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(persona_id, entity_kind, entity_id) DO UPDATE SET + fit_score = excluded.fit_score, + hard_filter_pass = excluded.hard_filter_pass, + components_json = excluded.components_json, + -- Drop the cached AI explanation when the new score falls below + -- the explanation threshold so we don't keep stale "why this + -- fits" rationale next to a low score. When the new score is + -- still high but no fresh explanation was generated this pass + -- (e.g. budget cap), keep the previous text — explanation_at + -- preserves the original timestamp so callers can detect age. + explanation = CASE + WHEN excluded.fit_score < 50 THEN NULL + WHEN excluded.explanation IS NOT NULL THEN excluded.explanation + ELSE persona_matches.explanation + END, + explanation_at = CASE + WHEN excluded.fit_score < 50 THEN NULL + WHEN excluded.explanation IS NOT NULL THEN excluded.explanation_at + ELSE persona_matches.explanation_at + END, + persona_modified_at = excluded.persona_modified_at, + entity_modified_at = excluded.entity_modified_at, + computed_at = excluded.computed_at`).bind(args.persona_id, args.entity_kind, args.entity_id, args.fit_score, args.hard_filter_pass, JSON.stringify(args.components), args.explanation, args.explanation ? now : null, args.persona_modified_at, args.entity_modified_at, now).run(); +} +export async function listMatches(env, personaId, opts) { + const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); + const offset = Math.max(0, opts.offset ?? 0); + const minScore = Math.max(0, opts.minScore ?? 0); + const kind = opts.kind ?? "account"; + if (kind === "account") { + const r = await env.DB.prepare(`SELECT pm.*, a.name AS entity_name, a.domain AS entity_domain, a.industry AS entity_industry, a.employees AS entity_employees, a.account_score AS entity_account_score + FROM persona_matches pm + JOIN accounts a ON a.id = pm.entity_id + WHERE pm.persona_id = ? AND pm.entity_kind = 'account' AND pm.fit_score >= ? + ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return r.results ?? []; + } + const r = await env.DB.prepare(`SELECT pm.*, b.name AS entity_name, b.title AS entity_title, b.seniority AS entity_seniority, b.account_id AS entity_account_id + FROM persona_matches pm + JOIN buyers b ON b.id = pm.entity_id + WHERE pm.persona_id = ? AND pm.entity_kind = 'buyer' AND pm.fit_score >= ? + ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return r.results ?? []; +} +export async function countMatches(env, personaId, minScore = 60) { + const r = await env.DB.prepare(`SELECT COUNT(*) AS c FROM persona_matches WHERE persona_id = ? AND fit_score >= ?`).bind(personaId, minScore).first(); + return r?.c ?? 0; +} +export async function deleteMatchesForPersona(env, personaId) { + await env.DB.prepare(`DELETE FROM persona_matches WHERE persona_id = ?`).bind(personaId).run(); +} +export async function listMatchesForEntity(env, entityKind, entityId) { + const r = await env.DB.prepare(`SELECT persona_id, fit_score FROM persona_matches + WHERE entity_kind = ? AND entity_id = ? + ORDER BY fit_score DESC`).bind(entityKind, entityId).all(); + return r.results ?? []; +} +// Task #58: surface persona-fit on the account/buyer detail pages. +// Joins persona_matches with personas so the dashboard can render a +// "Persona fit" panel without a second round-trip per row. Returns +// rows above `minScore` (default 50, matching the explanation cache +// floor in upsertMatch) sorted by score desc. Skips archived/deleted +// personas — matches against an archived persona are stale evidence. +export async function listMatchesForEntityWithDetails(env, entityKind, entityId, opts) { + const minScore = Math.max(0, opts?.minScore ?? 50); + const personaKind = opts?.personaKind ?? entityKind; + const r = await env.DB.prepare(`SELECT pm.persona_id, pm.fit_score, pm.hard_filter_pass, pm.components_json, + pm.explanation, pm.explanation_at, pm.computed_at, + p.name AS persona_name, p.kind AS persona_kind, + p.status AS persona_status, p.thesis AS persona_thesis + FROM persona_matches pm + JOIN personas p ON p.id = pm.persona_id + WHERE pm.entity_kind = ? AND pm.entity_id = ? + AND pm.fit_score >= ? + AND p.deleted_at IS NULL + AND p.status = 'active' + AND p.kind = ? + ORDER BY pm.fit_score DESC`).bind(entityKind, entityId, minScore, personaKind).all(); + return (r.results ?? []).map((row) => ({ + persona_id: row.persona_id, + persona_name: row.persona_name, + persona_kind: row.persona_kind, + persona_status: row.persona_status, + persona_thesis: row.persona_thesis, + fit_score: row.fit_score, + hard_filter_pass: row.hard_filter_pass, + components: row.components_json ? (() => { try { + return JSON.parse(row.components_json); + } + catch { + return null; + } })() : null, + explanation: row.explanation, + explanation_at: row.explanation_at, + computed_at: row.computed_at, + })); +} +// ----- entity fact loaders shared by scorer + workflow +export async function loadAccountFacts(env, accountId) { + const a = await env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id = ?`).bind(accountId).first(); + if (!a) + return null; + const tech = await env.DB.prepare(`SELECT vendor FROM account_tech WHERE account_id = ?`).bind(accountId).all(); + const sigs = await env.DB.prepare(`SELECT kind, weight, confidence, occurred_at FROM signals WHERE account_id = ? ORDER BY occurred_at DESC LIMIT 200`).bind(accountId).all(); + const buyers = await env.DB.prepare(`SELECT id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id = ? LIMIT 50`).bind(accountId).all(); + const facts = { + status: a.status, + domain: a.domain, + hq_country_iso2: a.hq_country_iso2, + size_band: a.size_band, + employees: a.employees, + industry: a.industry, + industries: parseJsonArr(a.industries_json), + funding_stage: a.funding_stage, + techs: (tech.results ?? []).map((t) => t.vendor), + signals: sigs.results ?? [], + buyers: (buyers.results ?? []).map((b) => ({ + account: null, title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + })), + last_modified: a.updated_at, + }; + return { name: a.name, facts, last_modified: a.updated_at }; +} +export async function loadBuyerFacts(env, buyerId) { + const b = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id = ?`).bind(buyerId).first(); + if (!b) + return null; + const acct = await loadAccountFacts(env, b.account_id); + const facts = { + account: acct?.facts ?? null, + title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + }; + return { name: b.name ?? b.title ?? buyerId, facts, last_modified: b.updated_at, account_id: b.account_id }; +} +// Bulk loader for AccountFacts. Issues 4 set-based queries (accounts + +// account_tech + signals + buyers, each with WHERE id IN (?...)) and +// stitches them together. Used by rescorePersonaFull and the +// /preview endpoint to avoid N round-trips over the D1 binding. +export async function loadAccountFactsBulk(env, ids) { + const out = new Map(); + if (!ids.length) + return out; + const ph = ids.map(() => "?").join(","); + const [a, tech, sigs, buyers] = await Promise.all([ + env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id IN (${ph})`).bind(...ids).all(), + env.DB.prepare(`SELECT account_id, vendor FROM account_tech WHERE account_id IN (${ph})`).bind(...ids).all(), + env.DB.prepare(`SELECT account_id, kind, weight, confidence, occurred_at FROM signals WHERE account_id IN (${ph}) ORDER BY occurred_at DESC`).bind(...ids).all(), + env.DB.prepare(`SELECT id, account_id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id IN (${ph})`).bind(...ids).all(), + ]); + const techByAcct = new Map(); + for (const t of tech.results ?? []) { + const arr = techByAcct.get(t.account_id) ?? []; + arr.push(t.vendor); + techByAcct.set(t.account_id, arr); + } + const sigByAcct = new Map(); + for (const s of sigs.results ?? []) { + const arr = sigByAcct.get(s.account_id) ?? []; + if (arr.length < 200) + arr.push({ kind: s.kind, weight: s.weight, confidence: s.confidence, occurred_at: s.occurred_at }); + sigByAcct.set(s.account_id, arr); + } + const buyByAcct = new Map(); + for (const b of buyers.results ?? []) { + const arr = buyByAcct.get(b.account_id) ?? []; + if (arr.length < 50) + arr.push({ account: null, title: b.title, seniority: b.seniority, department: b.department, is_decision_maker: b.is_decision_maker, last_modified: b.updated_at }); + buyByAcct.set(b.account_id, arr); + } + for (const row of a.results ?? []) { + const facts = { + status: row.status, domain: row.domain, hq_country_iso2: row.hq_country_iso2, + size_band: row.size_band, employees: row.employees, industry: row.industry, + industries: parseJsonArr(row.industries_json), funding_stage: row.funding_stage, + techs: techByAcct.get(row.id) ?? [], + signals: sigByAcct.get(row.id) ?? [], + buyers: buyByAcct.get(row.id) ?? [], + last_modified: row.updated_at, + }; + out.set(row.id, { name: row.name, facts, last_modified: row.updated_at }); + } + return out; +} +// Bulk loader for BuyerFacts. Issues 1 query for the buyers + delegates +// to loadAccountFactsBulk for parent accounts (one round-trip via IN). +export async function loadBuyerFactsBulk(env, ids) { + const out = new Map(); + if (!ids.length) + return out; + const ph = ids.map(() => "?").join(","); + const r = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id IN (${ph})`).bind(...ids).all(); + const buyerRows = r.results ?? []; + const acctIds = Array.from(new Set(buyerRows.map((b) => b.account_id))); + const acctFacts = await loadAccountFactsBulk(env, acctIds); + for (const b of buyerRows) { + const acct = acctFacts.get(b.account_id); + const facts = { + account: acct?.facts ?? null, title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + }; + out.set(b.id, { name: b.name ?? b.title ?? b.id, facts, last_modified: b.updated_at, account_id: b.account_id }); + } + return out; +} +// Bulk writeback: recompute max active-persona fit_score for each id +// in one aggregate query, then UPDATE in one statement per kind. Used +// at the end of each rescore batch instead of N per-row writebacks. +export async function bulkWriteBackFit(env, kind, ids) { + if (!ids.length) + return; + const ph = ids.map(() => "?").join(","); + const rows = await env.DB.prepare(`SELECT pm.entity_id AS id, MAX(pm.fit_score) AS m + FROM persona_matches pm + JOIN personas p ON p.id = pm.persona_id + WHERE pm.entity_kind = ? AND pm.entity_id IN (${ph}) + AND p.status = 'active' AND p.deleted_at IS NULL + GROUP BY pm.entity_id`).bind(kind, ...ids).all(); + const maxById = new Map(); + for (const r of rows.results ?? []) + maxById.set(r.id, r.m ?? 0); + // Issue updates as a batch (each binds its own params; D1 batches + // these into one HTTP round-trip via the binding's batch() API). + const stmts = ids.map((id) => { + const m = maxById.get(id) ?? 0; + if (kind === "account") { + return env.DB.prepare(`UPDATE accounts SET fit_score = ?, account_score = ROUND((0.6 * intent_score) + (0.4 * ?), 2) WHERE id = ?`).bind(m, m, id); + } + return env.DB.prepare(`UPDATE buyers SET fit_score = ? WHERE id = ?`).bind(m, id); + }); + await env.DB.batch(stmts); +} +// Materialize a compact "facts" object for the AI explainer. +export function summarizeAccountForExplanation(name, f) { + return { + name, + domain: f.domain, + industry: f.industry, + employees: f.employees, + size_band: f.size_band, + country: f.hq_country_iso2, + funding_stage: f.funding_stage, + top_techs: f.techs.slice(0, 8), + top_signals: f.signals.slice(0, 5).map((s) => ({ kind: s.kind, weight: s.weight, occurred_at: s.occurred_at })), + top_buyer: f.buyers[0] ? { title: f.buyers[0].title, seniority: f.buyers[0].seniority } : null, + }; +} +export function summarizeBuyerForExplanation(name, f) { + return { + name, + title: f.title, + seniority: f.seniority, + department: f.department, + is_decision_maker: !!f.is_decision_maker, + account: f.account ? summarizeAccountForExplanation(name, f.account) : null, + }; +} diff --git a/apps/worker/test-dist-q/personas/score.js b/apps/worker/test-dist-q/personas/score.js new file mode 100644 index 00000000..4cd5e110 --- /dev/null +++ b/apps/worker/test-dist-q/personas/score.js @@ -0,0 +1,281 @@ +// Task #46: deterministic persona scoring. +// +// fit_score = clamp(0..100, recency_boost * Σ w_i * c_i) when hard +// filters pass, else 0. Components are each 0..100. semantic_fit comes +// from cosine similarity against the entity's existing embedding (we +// pass it in pre-computed); the rest are computed here from row data. +export const DEFAULT_WEIGHTS_ACCOUNT = { + size: 0.10, + geo: 0.10, + industry: 0.20, + tech: 0.10, + signal: 0.20, + buyer: 0.10, + semantic: 0.20, +}; +export const DEFAULT_WEIGHTS_BUYER = { + size: 0.05, + geo: 0.05, + industry: 0.15, + tech: 0.05, + signal: 0.10, + buyer: 0.45, + semantic: 0.15, +}; +const DAY = 86_400_000; +function lc(s) { return (s ?? "").toLowerCase().trim(); } +function clamp(n, lo, hi) { return Math.max(lo, Math.min(hi, n)); } +function scoreSize(p, emp, band) { + if (p.size_min == null && p.size_max == null && p.size_bands.length === 0) + return 50; + if (band && p.size_bands.length && p.size_bands.includes(band)) + return 100; + if (emp == null) + return 25; + const lo = p.size_min ?? 0; + const hi = p.size_max ?? Number.MAX_SAFE_INTEGER; + if (emp >= lo && emp <= hi) + return 100; + // Soft penalty: 25% per order-of-magnitude away. + const ratio = emp < lo ? lo / Math.max(1, emp) : emp / hi; + const decades = Math.log10(ratio); + return Math.max(0, 100 - Math.round(decades * 60)); +} +function scoreGeo(p, iso) { + if (!p.geos.length) + return 50; + if (!iso) + return 20; + const i = lc(iso); + if (p.geos.includes(i)) + return 100; + // Region bucketing + const REGION = { + emea: ["gb", "ie", "fr", "de", "es", "it", "nl", "be", "se", "no", "fi", "dk", "pt", "pl", "ch", "at"], + apac: ["jp", "sg", "au", "nz", "kr", "in", "hk", "tw", "my", "id", "ph", "th", "vn"], + latam: ["br", "mx", "ar", "cl", "co", "pe", "uy"], + africa: ["za", "ng", "ke", "eg", "ma"], + }; + for (const slug of p.geos) { + const ctry = REGION[slug]; + if (ctry && ctry.includes(i)) + return 80; + if (slug === "global") + return 60; + } + return 0; +} +function scoreIndustry(p, primary, all) { + if (!p.industries.length) + return 50; + const want = new Set(p.industries.map(lc)); + if (primary && want.has(lc(primary))) + return 100; + for (const i of all.map(lc)) + if (want.has(i)) + return 80; + return 0; +} +function scoreTech(p, techs) { + const have = new Set(techs.map(lc)); + if (p.techs_excluded.some((t) => have.has(lc(t)))) + return { score: 0, pass: false }; + if (p.techs_required.length) { + const all = p.techs_required.every((t) => have.has(lc(t))); + if (!all) + return { score: 0, pass: false }; + } + if (!p.techs_preferred.length && !p.techs_required.length) + return { score: 50, pass: true }; + const matched = p.techs_preferred.filter((t) => have.has(lc(t))).length; + const ratio = p.techs_preferred.length ? matched / p.techs_preferred.length : 1; + return { score: Math.round(60 + 40 * ratio), pass: true }; +} +function scoreSignals(p, sigs) { + if (!sigs.length) + return 0; + const want = new Set(p.signal_kinds.map(lc)); + const now = Date.now(); + let totalCredit = 0; + for (const s of sigs) { + const t = Date.parse(s.occurred_at); + const age = Number.isFinite(t) ? Math.max(0, (now - t) / DAY) : 365; + const decay = Math.exp(-age / 30); // half-life-ish + const w = (s.weight ?? 0) * (s.confidence ?? 1); + const kindMul = want.size === 0 ? 0.5 : want.has(lc(s.kind)) ? 1.0 : 0.25; + totalCredit += w * decay * kindMul; + } + return Math.round(100 * (1 - Math.exp(-totalCredit / 15))); +} +function scoreBuyer(p, b) { + let s = 0; + let denom = 0; + if (p.buyer_titles.length) { + denom += 50; + const t = lc(b.title); + if (t && p.buyer_titles.some((x) => t.includes(lc(x)))) + s += 50; + } + if (p.buyer_seniority.length) { + denom += 30; + if (b.seniority && p.buyer_seniority.includes(lc(b.seniority))) + s += 30; + } + if (p.buyer_departments.length) { + denom += 20; + if (b.department && p.buyer_departments.includes(lc(b.department))) + s += 20; + } + if (denom === 0) + return 50; + // Decision-maker bonus + if (b.is_decision_maker) + s = Math.min(denom, s + 5); + return Math.round((s / denom) * 100); +} +function scoreBuyersForAccount(p, buyers) { + if (!buyers.length) + return p.buyer_titles.length || p.buyer_seniority.length ? 0 : 50; + let best = 0; + for (const b of buyers) + best = Math.max(best, scoreBuyer(p, b)); + return best; +} +export function checkHardFilters(p, account, buyer) { + const reasons = []; + const f = p.hard_filters || {}; + const acc = account ?? buyer?.account ?? null; + if (f.require_domain && acc && !acc.domain) { + reasons.push("missing_domain"); + return { pass: false, reasons }; + } + if (Array.isArray(f.statuses_in) && acc && !f.statuses_in.map(lc).includes(lc(acc.status))) { + reasons.push(`status_not_in:${acc.status}`); + return { pass: false, reasons }; + } + if (Array.isArray(f.exclude_country_iso2) && acc && acc.hq_country_iso2 && f.exclude_country_iso2.map(lc).includes(lc(acc.hq_country_iso2))) { + reasons.push(`country_excluded:${acc.hq_country_iso2}`); + return { pass: false, reasons }; + } + if (Array.isArray(f.country_iso2_in) && acc && (!acc.hq_country_iso2 || !f.country_iso2_in.map(lc).includes(lc(acc.hq_country_iso2)))) { + reasons.push("country_not_in"); + return { pass: false, reasons }; + } + if (Array.isArray(f.funding_stage_in) && acc && (!acc.funding_stage || !f.funding_stage_in.map(lc).includes(lc(acc.funding_stage)))) { + reasons.push("funding_stage_not_in"); + return { pass: false, reasons }; + } + if (f.is_decision_maker && buyer && !buyer.is_decision_maker) { + reasons.push("not_decision_maker"); + return { pass: false, reasons }; + } + return { pass: true, reasons }; +} +export function recencyBoost(p, lastModifiedISO) { + // Override wins; otherwise small boost (max 1.15x) for entities updated + // within the last 7 days. Capped 1.0..1.2 per spec. + if (p.recency_boost && p.recency_boost > 0) + return clamp(p.recency_boost, 1.0, 1.2); + if (!lastModifiedISO) + return 1.0; + const t = Date.parse(lastModifiedISO); + if (!Number.isFinite(t)) + return 1.0; + const ageDays = Math.max(0, (Date.now() - t) / DAY); + if (ageDays <= 7) + return 1.15; + if (ageDays <= 30) + return 1.05; + return 1.0; +} +export function scoreEntity(p, ctx) { + const reasons = []; + const hf = checkHardFilters(p, ctx.account, ctx.buyer); + if (!hf.pass) { + return { + fit_score: 0, + components: { + hard_filter_pass: 0, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, + signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, + weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, + reasons: hf.reasons, + }, + }; + } + const acc = ctx.account ?? ctx.buyer?.account ?? null; + const tech = scoreTech(p, acc?.techs ?? []); + if (!tech.pass) { + reasons.push("tech_excluded_or_missing_required"); + return { + fit_score: 0, + components: { + hard_filter_pass: 1, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, + signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, + weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, + reasons, + }, + }; + } + const size = scoreSize(p, acc?.employees ?? null, acc?.size_band ?? null); + const geo = scoreGeo(p, acc?.hq_country_iso2 ?? null); + const industry = scoreIndustry(p, acc?.industry ?? null, acc?.industries ?? []); + const signal = scoreSignals(p, acc?.signals ?? []); + const buyerScore = ctx.buyer ? scoreBuyer(p, ctx.buyer) : scoreBuyersForAccount(p, acc?.buyers ?? []); + const semCos = typeof ctx.semanticCosine === "number" ? ctx.semanticCosine : null; + const semantic = semCos == null + ? 50 + : semCos < (p.semantic_fit_threshold ?? 0.55) ? 0 : Math.round(clamp((semCos - 0.4) / 0.5, 0, 1) * 100); + const w = { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }; + const sumW = (w.size ?? 0) + (w.geo ?? 0) + (w.industry ?? 0) + (w.tech ?? 0) + (w.signal ?? 0) + (w.buyer ?? 0) + (w.semantic ?? 0); + const norm = sumW > 0 ? sumW : 1; + const blended = ((size * (w.size ?? 0)) + + (geo * (w.geo ?? 0)) + + (industry * (w.industry ?? 0)) + + (tech.score * (w.tech ?? 0)) + + (signal * (w.signal ?? 0)) + + (buyerScore * (w.buyer ?? 0)) + + (semantic * (w.semantic ?? 0))) / norm; + const boost = recencyBoost(p, ctx.buyer?.last_modified ?? acc?.last_modified ?? null); + const fit = Math.round(clamp(blended * boost, 0, 100)); + return { + fit_score: fit, + components: { + hard_filter_pass: 1, + size_fit: size, + geo_fit: geo, + industry_fit: industry, + tech_fit: tech.score, + signal_fit: signal, + buyer_fit: buyerScore, + semantic_fit: semantic, + recency_boost: boost, + weights: w, + reasons, + }, + }; +} +export function buildEmbeddingText(p) { + const parts = []; + parts.push(`Persona: ${p.name}`); + if (p.thesis) + parts.push(`Thesis: ${p.thesis}`); + if (p.industries.length) + parts.push(`Industries: ${p.industries.join(", ")}`); + if (p.geos.length) + parts.push(`Geos: ${p.geos.join(", ")}`); + if (p.size_min || p.size_max) + parts.push(`Size: ${p.size_min ?? "?"}–${p.size_max ?? "?"} FTE`); + if (p.techs_required.length) + parts.push(`Required tech: ${p.techs_required.join(", ")}`); + if (p.techs_preferred.length) + parts.push(`Preferred tech: ${p.techs_preferred.join(", ")}`); + if (p.signal_kinds.length) + parts.push(`Watch signals: ${p.signal_kinds.join(", ")}`); + if (p.buyer_titles.length) + parts.push(`Buyer titles: ${p.buyer_titles.join(", ")}`); + if (p.buyer_seniority.length) + parts.push(`Buyer seniority: ${p.buyer_seniority.join(", ")}`); + if (p.buyer_departments.length) + parts.push(`Buyer departments: ${p.buyer_departments.join(", ")}`); + return parts.join(" | "); +} diff --git a/apps/worker/test-dist-q/scraper/firms_upsert.js b/apps/worker/test-dist-q/scraper/firms_upsert.js new file mode 100644 index 00000000..5dbae1ad --- /dev/null +++ b/apps/worker/test-dist-q/scraper/firms_upsert.js @@ -0,0 +1,267 @@ +import { extractDomain } from "./normalize"; +import { syncFirmToEntity } from "../entities/dualwrite"; +const SCALAR_FIELDS = [ + "legal_name", "kind", "website", "logo_url", + "hq_country_iso2", "hq_region", "hq_city", + "thesis", "check_size_min_usd", "check_size_max_usd", "check_size_typical_usd", + "aum_usd", "fund_count", "current_fund_name", "current_fund_size_usd", + "lead_or_co", "portfolio_count", "founded_year", "team_size", + "linkedin_url", "crunchbase_url", "twitter_handle", + "signal_nfx_url", "openvc_url", "pitchbook_url", + "contact_email", "submission_url", +]; +// `source_url` is intentionally excluded from SCALAR_FIELDS — Task #1 +// requires that re-imports from different Folk shares union the +// provenance URLs at the firm row level rather than fill-if-empty. The +// merge path below comma-joins distinct values (mirrors `imported_from`). +const ARRAY_FIELDS = [ + { key: "geo_focus", column: "geo_focus_json" }, + { key: "stages", column: "stages_json" }, + { key: "sectors", column: "sectors_json" }, + { key: "notable_investments", column: "notable_investments_json" }, +]; +export async function upsertFirm(env, candidate, importedFrom, +/** + * Task #1: optional dual-write provenance override. Folk-share imports + * pass `{ source: 'folk_share', sourceKind: 'import' }` so the firm's + * unified-graph facts carry import provenance instead of the default + * `source_kind='scrape'`. + */ +importCtx) { + const name = candidate.name?.trim(); + if (!name) + throw new Error("upsertFirm: candidate.name required"); + const rawDomain = candidate.domain ?? deriveDomain(candidate.website); + const domain = rawDomain ? rawDomain.toLowerCase().trim() : null; + if (domain) + candidate.domain = domain; + // Quality gate: require name + (domain OR website). Without either, + // dedupe is impossible and reruns would create endless duplicates. + if (!domain && !candidate.website) { + throw new Error("upsertFirm: candidate must have domain or website"); + } + const lname = name.toLowerCase(); + // Dedupe lookup. Match on the effective domain (stored.domain coalesced + // with the parsed hostname of stored.website) so a rerun that supplies + // a domain still merges with a row that originally only had a website. + const candidates = await env.DB.prepare("SELECT * FROM firms WHERE lower(name) = ? LIMIT 50").bind(lname).all(); + const rows = candidates.results ?? []; + let existing = null; + for (const r of rows) { + const storedDomain = r.domain ?? deriveDomain(r.website ?? undefined); + if (domain && storedDomain && storedDomain.toLowerCase() === domain) { + existing = r; + break; + } + if (!domain && !storedDomain) { + existing = r; + break; + } + } + const result = existing + ? await mergeInto(env, existing, candidate, importedFrom) + : await insertNew(env, candidate, domain, importedFrom); + // Task #4: dual-write into the unified entity graph (best-effort — + // never block the legacy firm-list importer on a unified-model error). + try { + await syncFirmToEntity(env, { + id: result.firmId, + name: candidate.name, + legal_name: candidate.legal_name ?? null, + website: result.website, + domain: result.domain, + hq_country_iso2: candidate.hq_country_iso2 ?? null, + hq_region: candidate.hq_region ?? null, + hq_city: candidate.hq_city ?? null, + check_size_min_usd: candidate.check_size_min_usd ?? null, + check_size_max_usd: candidate.check_size_max_usd ?? null, + check_size_typical_usd: candidate.check_size_typical_usd ?? null, + thesis: candidate.thesis ?? null, + linkedin_url: candidate.linkedin_url ?? null, + crunchbase_url: candidate.crunchbase_url ?? null, + twitter_handle: candidate.twitter_handle ?? null, + contact_email: candidate.contact_email ?? null, + sectors_json: candidate.sectors ? JSON.stringify(candidate.sectors) : null, + stages_json: candidate.stages ? JSON.stringify(candidate.stages) : null, + geo_focus_json: candidate.geo_focus ? JSON.stringify(candidate.geo_focus) : null, + kind: candidate.kind ?? null, + }, importCtx?.source ?? importedFrom, importCtx?.sourceKind ?? "scrape"); + } + catch (e) { + console.warn("dualwrite syncFirmToEntity failed", result.firmId, e.message); + } + // Role inference now runs centrally inside syncFirmToEntity (the + // unified entity write path), so we no longer call it here. + return result; +} +async function insertNew(env, c, domain, importedFrom) { + const slug = await pickUniqueSlug(env, c.name, domain); + const cols = ["name", "slug", "domain", "imported_from", "last_modified"]; + const vals = [c.name.trim(), slug, domain, importedFrom, new Date().toISOString()]; + for (const f of SCALAR_FIELDS) { + const v = c[f]; + if (v != null && v !== "") { + cols.push(f); + vals.push(v); + } + } + for (const { key, column } of ARRAY_FIELDS) { + const v = c[key]; + if (v && v.length) { + cols.push(column); + vals.push(JSON.stringify(uniqStringArray(v))); + } + } + if (c.source_url) { + cols.push("source_url"); + vals.push(c.source_url); + } + if (c.socials) { + cols.push("socials_json"); + vals.push(JSON.stringify(c.socials)); + } + if (c.notes) { + cols.push("notes"); + vals.push(c.notes); + } + const placeholders = cols.map(() => "?").join(","); + const r = await env.DB.prepare(`INSERT INTO firms (${cols.join(",")}) VALUES (${placeholders})`).bind(...vals).run(); + const firmId = Number(r.meta.last_row_id); + return { firmId, action: "created", website: c.website ?? null, domain }; +} +async function mergeInto(env, existing, c, importedFrom) { + const sets = []; + const binds = []; + // Scalar: only fill missing values; existing non-null values win. + for (const f of SCALAR_FIELDS) { + const newVal = c[f]; + if (newVal == null || newVal === "") + continue; + if (existing[f] == null || existing[f] === "") { + sets.push(`${f} = ?`); + binds.push(newVal); + } + } + // Array fields: set-union of existing JSON array + new entries. + for (const { key, column } of ARRAY_FIELDS) { + const incoming = c[key]; + if (!incoming || !incoming.length) + continue; + const existingArr = parseJsonArray(existing[column]); + const merged = uniqStringArray([...existingArr, ...incoming]); + if (merged.length !== existingArr.length) { + sets.push(`${column} = ?`); + binds.push(JSON.stringify(merged)); + } + } + if (c.socials) { + const existingSocials = parseJsonObject(existing.socials_json); + const merged = { ...existingSocials, ...c.socials }; + if (Object.keys(merged).length !== Object.keys(existingSocials).length) { + sets.push("socials_json = ?"); + binds.push(JSON.stringify(merged)); + } + } + if (c.notes && !existing.notes) { + sets.push("notes = ?"); + binds.push(c.notes); + } + // Track every distinct origin. + const importedFromExisting = existing.imported_from ?? ""; + if (!importedFromExisting.split(",").includes(importedFrom)) { + sets.push("imported_from = ?"); + binds.push(importedFromExisting ? `${importedFromExisting},${importedFrom}` : importedFrom); + } + // Task #1: union source_url across re-imports. Folk shares (Top-300, + // FR VCs, etc.) each have their own share URL; re-importing the same + // firm from a second share must preserve evidence of both shares + // rather than fill-if-empty (which would silently drop the second + // URL). Mirrors the imported_from comma-join pattern above; the + // unified graph still gets one channel/fact per share via dualwrite. + const newSourceUrl = c.source_url; + if (typeof newSourceUrl === "string" && newSourceUrl) { + const existingSourceUrl = existing.source_url ?? ""; + const parts = existingSourceUrl ? existingSourceUrl.split(",").map((s) => s.trim()).filter(Boolean) : []; + if (!parts.includes(newSourceUrl)) { + parts.push(newSourceUrl); + sets.push("source_url = ?"); + binds.push(parts.join(",")); + } + } + // Always bump last_modified on every dedupe hit — even when no field + // deltas applied — so reruns leave a verifiable timestamp trail. + const action = sets.length ? "updated" : "unchanged"; + sets.push("last_modified = ?"); + binds.push(new Date().toISOString()); + binds.push(existing.id); + await env.DB.prepare(`UPDATE firms SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); + const persistedWebsite = existing.website ?? c.website ?? null; + const persistedDomain = existing.domain ?? deriveDomain(persistedWebsite); + return { firmId: existing.id, action, website: persistedWebsite, domain: persistedDomain }; +} +function deriveDomain(website) { + if (!website) + return null; + return extractDomain(website) || null; +} +function slugify(s) { + return s.toLowerCase().normalize("NFKD").replace(/[^\w]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80); +} +async function pickUniqueSlug(env, name, domain) { + const base = slugify(name) || "firm"; + let candidate = base; + let row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); + if (!row) + return candidate; + if (domain) { + candidate = `${base}-${slugify(domain)}`; + row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); + if (!row) + return candidate; + } + for (let i = 2; i < 100; i++) { + const c = `${base}-${i}`; + row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(c).first(); + if (!row) + return c; + } + // Last resort: random suffix. + return `${base}-${Math.random().toString(36).slice(2, 8)}`; +} +function parseJsonArray(raw) { + if (!raw) + return []; + try { + const v = JSON.parse(raw); + return Array.isArray(v) ? v.map((x) => String(x)) : []; + } + catch { + return []; + } +} +function parseJsonObject(raw) { + if (!raw) + return {}; + try { + const v = JSON.parse(raw); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +function uniqStringArray(arr) { + const seen = new Set(); + const out = []; + for (const s of arr) { + const k = String(s).trim(); + if (!k) + continue; + const lk = k.toLowerCase(); + if (seen.has(lk)) + continue; + seen.add(lk); + out.push(k); + } + return out; +} diff --git a/apps/worker/test-dist-q/scraper/normalize.js b/apps/worker/test-dist-q/scraper/normalize.js new file mode 100644 index 00000000..5f9c29b0 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/normalize.js @@ -0,0 +1,104 @@ +// Normalization helpers used by the parsers and dedupe key generators. +const EMAIL_RE = /^[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}$/i; +const PHONE_DIGITS_RE = /[^\d+]/g; +export function normalizeEmail(raw) { + if (!raw) + return null; + const s = raw.trim().toLowerCase(); + if (!EMAIL_RE.test(s)) + return null; + return s; +} +/** + * Email key for dedupe purposes only — strips +tags and lowercases the domain. + * The displayed email uses normalizeEmail (preserves the +tag). + */ +export function emailDedupeKey(email) { + const e = normalizeEmail(email); + if (!e) + return null; + const [localPart, domain] = e.split("@"); + const stripped = localPart.split("+")[0]; + return `${stripped}@${domain.toLowerCase()}`; +} +/** + * Best-effort E.164 normalization. We only accept numbers that already include + * a leading + or are clearly internationalizable (10–15 digits). Otherwise null. + */ +export function normalizePhoneE164(raw) { + if (!raw) + return null; + const cleaned = raw.replace(PHONE_DIGITS_RE, ""); + if (!cleaned) + return null; + if (cleaned.startsWith("+")) { + const digits = cleaned.slice(1); + if (digits.length < 7 || digits.length > 15) + return null; + return `+${digits}`; + } + // No country code → don't guess; return null so we don't pollute dedupe keys. + if (cleaned.length === 10 || cleaned.length === 11) { + // Common case: US/Canada 10 digits or 1+10 digits. + const d = cleaned.length === 10 ? `1${cleaned}` : cleaned; + return `+${d}`; + } + return null; +} +export function canonicalizeLinkedinUrl(url) { + if (!url) + return null; + try { + const u = new URL(url); + const host = u.hostname.toLowerCase().replace(/^www\./, ""); + if (!host.endsWith("linkedin.com")) + return null; + // Trailing slash and query strip + const path = u.pathname.replace(/\/+$/, "").toLowerCase(); + if (!/^\/(in|company|school)\/[^\/]+/.test(path)) + return null; + return `https://www.linkedin.com${path}`; + } + catch { + return null; + } +} +export function countryNameToIso2(name) { + if (!name) + return null; + const s = name.trim().toLowerCase(); + // Tiny seed map; the full mapping lives with the taxonomies task. + const seed = { + usa: "US", + "united states": "US", + "united states of america": "US", + canada: "CA", + uk: "GB", + "united kingdom": "GB", + england: "GB", + france: "FR", + germany: "DE", + spain: "ES", + italy: "IT", + netherlands: "NL", + switzerland: "CH", + israel: "IL", + india: "IN", + china: "CN", + japan: "JP", + singapore: "SG", + australia: "AU", + brazil: "BR", + }; + if (s.length === 2) + return s.toUpperCase(); + return seed[s] ?? null; +} +export function extractDomain(url) { + try { + return new URL(url).hostname.toLowerCase().replace(/^www\./, ""); + } + catch { + return ""; + } +} diff --git a/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js b/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js new file mode 100644 index 00000000..cb0ff5c3 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js @@ -0,0 +1 @@ +export {}; diff --git a/apps/worker/test-dist-q/scraper/rateLimit.js b/apps/worker/test-dist-q/scraper/rateLimit.js new file mode 100644 index 00000000..ebd48fb8 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/rateLimit.js @@ -0,0 +1,53 @@ +// Per-host + global AI rate limiting (Task #25 step 6). +// +// Prefers the Cloudflare Rate Limiter binding (`RL_HOST`/`RL_AI`) when +// configured. Falls back to a KV-backed leaky-bucket counter on +// SCRAPE_CACHE so the worker keeps pacing itself even when the binding is +// missing in dev or before the namespace is provisioned. +const KV_PREFIX = "rl:"; +const HOST_LIMIT_PER_MIN = 60; +const AI_LIMIT_PER_MIN = 600; +const WINDOW_MS = 60_000; +async function kvLeakyBucket(kv, key, limit) { + if (!kv) + return true; + const raw = await kv.get(key); + const now = Date.now(); + let bucket = raw ? safeParse(raw) : { count: 0, window_start: now }; + if (now - bucket.window_start > WINDOW_MS) + bucket = { count: 0, window_start: now }; + if (bucket.count >= limit) + return false; + bucket.count += 1; + await kv.put(key, JSON.stringify(bucket), { expirationTtl: 120 }); + return true; +} +function safeParse(raw) { + try { + const v = JSON.parse(raw); + if (typeof v?.count === "number" && typeof v?.window_start === "number") + return v; + } + catch { /* swallow */ } + return { count: 0, window_start: Date.now() }; +} +export async function limitHost(env, host) { + if (env.RL_HOST) { + try { + const r = await env.RL_HOST.limit({ key: host }); + return r.success; + } + catch { /* fall through to KV */ } + } + return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}host:${host}`, HOST_LIMIT_PER_MIN); +} +export async function limitAi(env) { + if (env.RL_AI) { + try { + const r = await env.RL_AI.limit({ key: "global" }); + return r.success; + } + catch { /* fall through */ } + } + return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}ai:global`, AI_LIMIT_PER_MIN); +} diff --git a/apps/worker/test-dist-q/services/personaMatchTrigger.js b/apps/worker/test-dist-q/services/personaMatchTrigger.js new file mode 100644 index 00000000..16f48214 --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatchTrigger.js @@ -0,0 +1,59 @@ +// Task #8: per-entity persona-match refresh trigger. +// +// Called from entity write paths (insertFact, addCareerEntry) so a new +// job or relocation flows into persona candidate rankings within +// minutes, not at the next nightly cron. Debounced via KV so a burst +// of fact writes for the same entity only triggers one re-match. +// Predicates that materially affect a person-entity's persona score. +// Other predicates (donations, family ties, lifestyle, etc.) are +// ignored here so we don't dispatch on unrelated edits. +const RELEVANT_PREDICATES = new Set([ + "person.career", "person.title", "title", + "person.seniority", "person.department", + "person.location.country", "person.location.city", + "location.country", "location.city", + "employer", "person.employer", + // `employees` is the spelling that actually gets written; without it a + // fresh headcount fact never re-scored the entity it belongs to. + "employees", + "org.headcount", "org.employees", + "company.employees", "company.headcount", + "org.sector", "sector", + "org.stage", "stage", +]); +export function isRelevantPredicate(predicate) { + if (!predicate) + return false; + return RELEVANT_PREDICATES.has(predicate); +} +const DEBOUNCE_SECONDS = 300; // 5 minutes +export async function triggerEntityMatchRefresh(env, entityId) { + if (!entityId) + return; + // KV debounce — first write wins per 5min window. + try { + const kvKey = `pem:trigger:${entityId}`; + if (env.SESSIONS) { + const existing = await env.SESSIONS.get(kvKey); + if (existing) + return; + await env.SESSIONS.put(kvKey, "1", { expirationTtl: DEBOUNCE_SECONDS }); + } + } + catch (e) { + console.warn("triggerEntityMatchRefresh debounce check failed", entityId, e.message); + // Fall through — better to dispatch than miss the trigger. + } + // Dispatch the per-entity workflow; inline fallback runs the service. + try { + if (env.WF_PERSONA_MATCH_ENTITY) { + await env.WF_PERSONA_MATCH_ENTITY.create({ params: { entityId } }); + return; + } + const { scoreEntityAcrossPersonas } = await import("./personaMatching.js"); + await scoreEntityAcrossPersonas(env, entityId); + } + catch (e) { + console.warn("triggerEntityMatchRefresh dispatch failed", entityId, e.message); + } +} diff --git a/apps/worker/test-dist-q/services/personaMatching.js b/apps/worker/test-dist-q/services/personaMatching.js new file mode 100644 index 00000000..c2e893c1 --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatching.js @@ -0,0 +1,484 @@ +// Task #8: Real persona matching algorithm. +// +// Deterministic weighted scoring engine that ranks unified `u_entities` +// (person entities) against a persona. Each entity gets a score in +// [0,1] plus a transparent per-component breakdown so the dashboard +// can explain *why* an entity matched. +// +// Pure scoring primitives live in personaMatchingScorers.ts (no Env +// imports — unit-testable). This module orchestrates the D1 loads, +// the title embedding, and the upsert. +import { aiEmbed } from "../ai/extract"; +import { assertBudget } from "../ai/budget"; +import { getPersona } from "../personas/repo"; +import { DEFAULT_WEIGHTS, MODEL_VERSION, cosine, aggregate, buildRationale, extractTargets, scoreSeniority, scoreFunction, scoreIndustry, scoreCompanySize, scoreStage, scoreGeo, } from "./personaMatchingScorers"; +export { DEFAULT_WEIGHTS, MODEL_VERSION, extractTargets }; +// Task #3: structural-only fallback used when a kind plugin returns +// null (e.g. fund/company targets that have no per-entity scoring +// pipeline). Builds a properly-typed ComponentMap with zeroed +// components so downstream consumers (rationale builder, persistence) +// don't have to special-case the structural row. Replaces an earlier +// `as unknown as MatchResult` cast that bypassed the type system. +export function buildStructuralFallback(reason) { + const components = {}; + for (const key of Object.keys(DEFAULT_WEIGHTS)) { + components[key] = { value: 0, weight: 0, reason: "n/a (structural fallback)" }; + } + return { score: 0.5, components, rationale: reason }; +} +// --------------------------------------------------------------------------- +// Entity loader. +// --------------------------------------------------------------------------- +async function loadEmployerFacts(env, employerId) { + const sum = await env.DB.prepare(`SELECT display_name, country_iso2, sectors_csv, stages_csv FROM entity_summary WHERE entity_id = ?`).bind(employerId).first(); + // `employees` is first because it is the only one of these that anything + // writes: it is the predicate the registry declares + // (entities/profile-predicates.ts) and the one secEdgar/persist.ts and the + // account dual-write emit. The four `org.*` / `company.*` spellings below + // were the entire list, and no writer has ever produced one — so + // `employees` came back null for every entity and scoreCompanySize + // returned its "company size unknown" zero every time. With a weight of + // 0.10 that put a hard ceiling of 0.90 on every persona match, and made + // "company size unknown" a permanent line in the rationale the dashboard + // shows to explain why someone matched. The unused spellings are kept so a + // future writer picking one still resolves. + const hc = await env.DB.prepare(`SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('employees','org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(employerId).first(); + if (!sum && !hc) + return null; + return { + name: sum?.display_name ?? null, + country: sum?.country_iso2 ?? null, + sectors: sum?.sectors_csv ? sum.sectors_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], + stages: sum?.stages_csv ? sum.stages_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], + employees: hc?.value_number != null ? Math.round(hc.value_number) : null, + }; +} +async function loadEntityCoords(env, entityId) { + const r = await env.DB.prepare(`SELECT predicate, value_number FROM facts + WHERE entity_id = ? AND is_current = 1 + AND predicate IN ('person.location.lat','person.location.lng','geo.lat','geo.lng','location.lat','location.lng') + AND value_number IS NOT NULL`).bind(entityId).all(); + let lat = null; + let lng = null; + for (const row of r.results ?? []) { + if (lat == null && (row.predicate.endsWith(".lat") || row.predicate === "geo.lat")) + lat = row.value_number; + if (lng == null && (row.predicate.endsWith(".lng") || row.predicate === "geo.lng")) + lng = row.value_number; + } + return { lat, lng }; +} +export async function loadPersonEntity(env, entityId) { + const ent = await env.DB.prepare(`SELECT id, display_name, kind, status FROM u_entities WHERE id = ?`).bind(entityId).first(); + if (!ent) + return null; + if (ent.kind !== "person") + return null; + if (ent.status === "merged" || ent.status === "soft_deleted") + return null; + const sum = await env.DB.prepare(`SELECT country_iso2, region FROM entity_summary WHERE entity_id = ?`).bind(entityId).first(); + const career = await env.DB.prepare(`SELECT role_title, seniority, department, organization_entity_id, organization_name + FROM career_history + WHERE entity_id = ? + ORDER BY is_current DESC, COALESCE(ended_at, '9999') DESC, started_at DESC + LIMIT 1`).bind(entityId).first(); + let title = career?.role_title ?? null; + if (!title) { + const tf = await env.DB.prepare(`SELECT value_text FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('person.title','title') AND value_text IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(entityId).first(); + title = tf?.value_text ?? null; + } + let employer = null; + if (career?.organization_entity_id) { + try { + employer = await loadEmployerFacts(env, career.organization_entity_id); + } + catch { /* ignore */ } + } + const coords = await loadEntityCoords(env, entityId).catch(() => ({ lat: null, lng: null })); + return { + id: ent.id, + display_name: ent.display_name, + country_iso2: sum?.country_iso2 ?? null, + region: sum?.region ?? null, + lat: coords.lat, + lng: coords.lng, + title, + seniority: career?.seniority ?? null, + department: career?.department ?? null, + employer_entity_id: career?.organization_entity_id ?? null, + employer_name: employer?.name ?? career?.organization_name ?? null, + employer_country: employer?.country ?? null, + employer_sectors: employer?.sectors ?? [], + employer_stages: employer?.stages ?? [], + employer_employees: employer?.employees ?? null, + }; +} +// --------------------------------------------------------------------------- +// Title similarity (only DB/Env-touching scorer). +// +// Embeddings are cached in persona_title_embeddings + entity_title_embeddings +// keyed by content_hash so the hot path becomes a D1 lookup instead of an +// AI.embed call. AI.embed only fires on cache miss (new persona, new entity, +// or title text change). This mirrors the Vectorize precompute/reuse +// pattern from Task #7 personas while staying on D1. +// --------------------------------------------------------------------------- +async function sha256Hex(input) { + const buf = new TextEncoder().encode(input); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} +async function ensureTitleCacheTables(env) { + try { + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_title_embeddings (persona_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS entity_title_embeddings (entity_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); + } + catch { /* best-effort */ } +} +async function getOrEmbedTitle(env, scope, id, text) { + const hash = await sha256Hex(text); + const table = scope === "persona" ? "persona_title_embeddings" : "entity_title_embeddings"; + const idCol = scope === "persona" ? "persona_id" : "entity_id"; + try { + const row = await env.DB.prepare(`SELECT vector_json FROM ${table} WHERE ${idCol} = ? AND content_hash = ?`).bind(id, hash).first(); + if (row?.vector_json) { + try { + const v = JSON.parse(row.vector_json); + if (Array.isArray(v) && v.length) + return v; + } + catch { /* fall through to re-embed */ } + } + } + catch { + // Table missing — create it once and continue with embedding path. + await ensureTitleCacheTables(env); + } + if (!env.AI) + return null; + const vec = await aiEmbed(env, text); + if (vec && vec.length) { + try { + await env.DB.prepare(`INSERT INTO ${table} (${idCol}, content_hash, vector_json, updated_at) VALUES (?, ?, ?, datetime('now')) + ON CONFLICT(${idCol}) DO UPDATE SET content_hash=excluded.content_hash, vector_json=excluded.vector_json, updated_at=excluded.updated_at`).bind(id, hash, JSON.stringify(vec)).run(); + } + catch (e) { + console.warn("title embedding cache write failed", scope, id, e.message); + } + } + return vec; +} +async function titleSimilarity(env, personaId, personaText, entityId, entityTitle) { + if (!entityTitle || !personaText) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "missing title" }; + } + if (!env.AI) { + const ov = scoreFunction(entityTitle, [personaText]); + return { value: ov.value, weight: DEFAULT_WEIGHTS.title_sim, reason: `title token overlap (no embed): ${ov.reason}` }; + } + try { + const [pv, ev] = await Promise.all([ + getOrEmbedTitle(env, "persona", personaId, personaText), + getOrEmbedTitle(env, "entity", entityId, entityTitle), + ]); + if (!pv || !ev) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "embedding unavailable" }; + } + const c = cosine(pv, ev); + return { value: c, weight: DEFAULT_WEIGHTS.title_sim, reason: `title cosine ${c.toFixed(3)} ("${entityTitle}")`, data: { cosine: c } }; + } + catch (e) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: `embed error: ${e.message.slice(0, 60)}` }; + } +} +// --------------------------------------------------------------------------- +// Public API. +// --------------------------------------------------------------------------- +export async function scoreEntityForPersona(env, persona, entity) { + const targets = extractTargets(persona); + const [title_sim, seniority, fnc, industry, company_size, stage, geo] = await Promise.all([ + titleSimilarity(env, persona.id, targets.title_text || targets.titles.join(", "), entity.id, entity.title), + Promise.resolve(scoreSeniority(entity.seniority, targets.seniority)), + Promise.resolve(scoreFunction(entity.department, targets.functions)), + Promise.resolve(scoreIndustry(entity.employer_sectors, targets.industries)), + Promise.resolve(scoreCompanySize(entity.employer_employees, targets.size_min, targets.size_max)), + Promise.resolve(scoreStage(entity.employer_stages, targets.stages)), + Promise.resolve(scoreGeo({ + entityIso2: entity.country_iso2 ?? entity.employer_country, + entityLat: entity.lat, entityLng: entity.lng, + targets: targets.geos, + centerLat: targets.geo_center_lat, centerLng: targets.geo_center_lng, + radiusKm: targets.geo_radius_km, + })), + ]); + const components = { title_sim, seniority, function: fnc, industry, company_size, stage, geo }; + const score = aggregate(components); + const rationale = buildRationale(persona.name, entity.display_name, entity.employer_name, components, score); + return { score, components, rationale }; +} +export async function scoreEntity(env, personaId, entityId) { + const persona = await getPersona(env, personaId); + if (!persona) + return null; + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + // Task #2 budget gate: refuse if AI cap reached (title_sim uses AI.embed). + const b = await assertBudget(env, "ai"); + if (!b.ok) { + await recordMatchJob(env, "score_entity", "halted", { personaId, entityId, reason: b.reason }); + return null; + } + const result = await scoreEntityForPersona(env, persona, entity); + await upsertMatch(env, personaId, entityId, result); + return result; +} +// Task #8: durable job/error log so SLO violations are visible. Created +// on demand by triggering migrations; CREATE TABLE IF NOT EXISTS guards +// against pre-migration calls. +async function ensureJobsTable(env) { + try { + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_match_jobs ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, + status TEXT NOT NULL, + persona_id TEXT, + entity_id TEXT, + details_json TEXT, + created_at TEXT NOT NULL DEFAULT (datetime('now')) + )`).run(); + } + catch { /* best-effort */ } +} +export async function recordMatchJob(env, kind, status, details) { + await ensureJobsTable(env); + try { + await env.DB.prepare(`INSERT INTO persona_match_jobs (id, kind, status, persona_id, entity_id, details_json) + VALUES (?, ?, ?, ?, ?, ?)`).bind(crypto.randomUUID(), kind, status, details.personaId ?? null, details.entityId ?? null, JSON.stringify(details)).run(); + } + catch (e) { + console.warn("recordMatchJob failed", kind, status, e.message); + } +} +async function isCancelled(env, jobId) { + if (!jobId) + return false; + try { + const r = await env.DB.prepare("SELECT status FROM jobs WHERE id = ?").bind(jobId).first(); + return r?.status === "cancelled" || r?.status === "timed_out"; + } + catch { + return false; + } +} +export async function upsertMatch(env, personaId, entityId, result, opts = {}) { + const source = opts.source ?? "auto"; + const evidence = JSON.stringify({ + components: Object.fromEntries(Object.keys(result.components).map((k) => [k, { + value: Number(result.components[k].value.toFixed(4)), + weight: result.components[k].weight, + reason: result.components[k].reason, + }])), + rationale: result.rationale, + weights: DEFAULT_WEIGHTS, + version: MODEL_VERSION, + }); + if (source === "auto") { + await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) + VALUES (?, ?, ?, ?, 'auto', datetime('now'), ?) + ON CONFLICT(persona_id, entity_id) DO UPDATE SET + score = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.score ELSE excluded.score END, + match_evidence_json = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.match_evidence_json ELSE excluded.match_evidence_json END, + last_scored_at = excluded.last_scored_at, + model_version = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.model_version ELSE excluded.model_version END`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); + return; + } + await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) + VALUES (?, ?, ?, ?, 'manual', datetime('now'), ?) + ON CONFLICT(persona_id, entity_id) DO UPDATE SET + score = excluded.score, + match_evidence_json = excluded.match_evidence_json, + source = 'manual', + last_scored_at = excluded.last_scored_at, + model_version = excluded.model_version`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); +} +export async function scoreEntityAcrossPersonas(env, entityId, opts = {}) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return { scored: 0, errors: 0, halted: false }; + const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = 'active'`).all(); + let scored = 0; + let errors = 0; + let halted = false; + for (const p of r.results ?? []) { + // Task #2: budget + cancellation enforcement per item. + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "score_across_personas", "halted", { entityId, scored, errors, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "score_across_personas", "cancelled", { entityId, scored, errors }); + break; + } + try { + const res = await scoreEntityForPersona(env, p, entity); + await upsertMatch(env, p.id, entityId, res); + scored += 1; + } + catch (e) { + errors += 1; + console.warn("scoreEntityAcrossPersonas item failed", p.id, entityId, e.message); + } + } + return { scored, errors, halted }; +} +export async function scoreBatch(env, personaId, opts = {}) { + const persona = await getPersona(env, personaId); + if (!persona) + return { scored: 0, errors: 0, pages: 0, halted: false }; + const batchSize = Math.min(Math.max(1, opts.batchSize ?? 100), 500); + // maxEntities = null (default) means "process every active person + // entity" — the task requires create/edit dispatch covers all + // entities, not a hardcoded cap. Operators can pass a number when + // they want to bound a manual run. + const maxEntities = opts.maxEntities ?? null; + let offset = 0; + let scored = 0; + let errors = 0; + let pages = 0; + let halted = false; + for (;;) { + if (maxEntities != null && scored + errors >= maxEntities) + break; + // Task #2: budget + cancellation check per page (cheap, bounded). + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "score_batch", "halted", { personaId, scored, errors, pages, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "score_batch", "cancelled", { personaId, scored, errors, pages }); + break; + } + // Task #3: dispatch through the kind plugin so each persona kind + // selects its own candidate pool (e.g. investor_person filters + // entity_roles.role IN ('investor','vc','gp','partner_at_firm')). + // Note: explicit .js extension here is required by tsconfig.test.json's + // NodeNext moduleResolution. The wrangler build / typecheck doesn't + // care; this is purely to unblock `pnpm test`. + const { getPluginFor } = await import("./personas/kinds/index.js"); + const plugin = getPluginFor(persona.kind); + const filter = plugin.defaultEntityFilter(persona, { limit: batchSize, offset }); + const r = await env.DB.prepare(filter.sql).bind(...filter.binds).all(); + const ids = (r.results ?? []).map((x) => x.id); + if (!ids.length) + break; + pages += 1; + for (const id of ids) { + try { + // Task #3: delegate to the kind plugin so bespoke matchers + // (investor_firm structural, venture_partner subtype, etc.) + // get the chance to override scoring. The generic plugin's + // scoreEntity returns the person-graph score for person + // targets and null for fund/company targets — when null, we + // persist a deterministic structural-match row at score 50 + // so non-person kinds still surface candidates in the UI. + let res = await plugin.scoreEntity(env, persona, id); + if (!res) + res = buildStructuralFallback(plugin.explainMatch(id)); + await upsertMatch(env, personaId, id, res); + scored += 1; + } + catch (e) { + errors += 1; + console.warn("scoreBatch item failed", personaId, id, e.message); + } + } + if (ids.length < batchSize) + break; + offset += batchSize; + } + if (errors > 0 && !halted) { + await recordMatchJob(env, "score_batch", "ok", { personaId, scored, errors, pages }); + } + return { scored, errors, pages, halted }; +} +export async function refreshStaleMatches(env, opts = {}) { + const staleDays = Math.max(1, opts.staleDays ?? 30); + const limit = Math.min(Math.max(1, opts.limit ?? 500), 5000); + const r = await env.DB.prepare(`SELECT persona_id, entity_id FROM persona_entity_matches + WHERE source = 'auto' AND datetime(last_scored_at) < datetime('now', ?) + ORDER BY last_scored_at ASC LIMIT ?`).bind(`-${staleDays} days`, limit).all(); + let refreshed = 0; + let errors = 0; + let halted = false; + for (const row of r.results ?? []) { + // Task #2: per-item budget + cancellation gate. + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "refresh_stale", "halted", { refreshed, errors, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "refresh_stale", "cancelled", { refreshed, errors }); + break; + } + try { + const res = await scoreEntity(env, row.persona_id, row.entity_id); + if (res) + refreshed += 1; + } + catch (e) { + errors += 1; + console.warn("refreshStaleMatches item failed", row.persona_id, row.entity_id, e.message); + } + } + return { refreshed, errors, halted }; +} +export async function listCandidates(env, personaId, opts) { + const minScore = Math.max(0, Math.min(1, opts.minScore ?? 0)); + const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); + const offset = Math.max(0, opts.offset ?? 0); + const r = await env.DB.prepare(`SELECT pem.persona_id, pem.entity_id, pem.score, pem.source, pem.last_scored_at, pem.model_version, pem.match_evidence_json, + ue.display_name AS entity_name, ue.primary_domain AS entity_domain, + es.country_iso2 AS entity_country + FROM persona_entity_matches pem + JOIN u_entities ue ON ue.id = pem.entity_id + LEFT JOIN entity_summary es ON es.entity_id = pem.entity_id + WHERE pem.persona_id = ? AND pem.score >= ? + ORDER BY pem.score DESC, pem.last_scored_at DESC + LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return (r.results ?? []).map((row) => { + let components = {}; + let rationale = ""; + if (row.match_evidence_json) { + try { + const j = JSON.parse(row.match_evidence_json); + if (j.components) + components = j.components; + if (typeof j.rationale === "string") + rationale = j.rationale; + } + catch { /* ignore */ } + } + return { + persona_id: row.persona_id, + entity_id: row.entity_id, + score: row.score, + source: row.source, + last_scored_at: row.last_scored_at, + model_version: row.model_version, + components, + rationale, + entity_name: row.entity_name, + entity_domain: row.entity_domain, + entity_country: row.entity_country, + }; + }); +} diff --git a/apps/worker/test-dist-q/services/personaMatchingScorers.js b/apps/worker/test-dist-q/services/personaMatchingScorers.js new file mode 100644 index 00000000..0d68be9d --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatchingScorers.js @@ -0,0 +1,366 @@ +// Task #8: Pure scoring primitives for the persona ↔ entity matcher. +// +// This module is intentionally free of D1 / Env / AI imports so it can +// be unit-tested in node:test (see test/personaMatching.test.mjs). The +// orchestrator (services/personaMatching.ts) wires these primitives to +// the database, embeddings, and workflow dispatch. +export const MODEL_VERSION = "v1"; +export const DEFAULT_WEIGHTS = { + title_sim: 0.25, + seniority: 0.15, + function: 0.15, + industry: 0.15, + company_size: 0.10, + stage: 0.10, + geo: 0.10, +}; +const SENIORITY_LADDER = [ + "ic", "analyst", "associate", "manager", "principal", + "director", "vp", "svp", "cxo", "founder", "partner", +]; +const SENIORITY_INDEX = Object.fromEntries(SENIORITY_LADDER.map((s, i) => [s, i])); +const STAGE_LADDER = [ + "pre_seed", "seed", "series_a", "series_b", "series_c", + "series_d", "growth", "late", "public", +]; +const STAGE_INDEX = Object.fromEntries(STAGE_LADDER.map((s, i) => [s, i])); +const INDUSTRY_PARENTS = { + fintech: ["finance"], insurtech: ["finance"], wealthtech: ["finance"], + proptech: ["realestate"], regtech: ["finance", "compliance"], + edtech: ["education"], healthtech: ["healthcare"], + biotech: ["healthcare", "lifesciences"], medtech: ["healthcare"], + cleantech: ["energy"], climatetech: ["energy"], + saas: ["software"], devtools: ["software"], paas: ["software"], + martech: ["marketing"], adtech: ["marketing"], + agtech: ["agriculture"], foodtech: ["food"], + legaltech: ["legal"], hrtech: ["hr"], +}; +const CONTINENT = { + us: "na", ca: "na", mx: "na", + gb: "eu", de: "eu", fr: "eu", es: "eu", it: "eu", nl: "eu", se: "eu", ch: "eu", ie: "eu", pl: "eu", pt: "eu", be: "eu", at: "eu", dk: "eu", no: "eu", fi: "eu", + cn: "as", jp: "as", in: "as", sg: "as", kr: "as", hk: "as", il: "as", ae: "as", + br: "sa", ar: "sa", cl: "sa", co: "sa", + au: "oc", nz: "oc", + za: "af", ng: "af", ke: "af", eg: "af", +}; +// Rough country centroids (lat, lng) used as a fallback when an entity +// has an ISO2 country but no precise coordinates. Only the most common +// targets are populated; absent entries skip the Haversine path and +// fall back to ISO2 + continent. +const COUNTRY_CENTROID = { + us: [39.8, -98.6], ca: [56.1, -106.3], mx: [23.6, -102.5], + gb: [54.0, -2.0], de: [51.2, 10.4], fr: [46.6, 2.2], + es: [40.4, -3.7], it: [41.9, 12.5], nl: [52.1, 5.3], + se: [60.1, 18.6], ch: [46.8, 8.2], ie: [53.1, -7.7], + cn: [35.9, 104.2], jp: [36.2, 138.2], in: [20.6, 78.9], + sg: [1.35, 103.8], kr: [35.9, 127.8], il: [31.0, 34.9], + br: [-14.2, -51.9], au: [-25.3, 133.8], nz: [-40.9, 174.9], + za: [-30.6, 22.9], ng: [9.1, 8.7], +}; +// --------------------------------------------------------------------------- +// Cosine similarity (exported for the embedding-driven title_sim). +// --------------------------------------------------------------------------- +export function cosine(a, b) { + if (a.length !== b.length || !a.length) + return 0; + let dot = 0, na = 0, nb = 0; + for (let i = 0; i < a.length; i++) { + dot += a[i] * b[i]; + na += a[i] * a[i]; + nb += b[i] * b[i]; + } + if (!na || !nb) + return 0; + return Math.max(0, dot / (Math.sqrt(na) * Math.sqrt(nb))); +} +// --------------------------------------------------------------------------- +// Component scorers (each returns a value in [0, 1]). +// --------------------------------------------------------------------------- +export function scoreSeniority(entity, targets) { + if (!entity || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: "no seniority data" }; + } + const e = entity.toLowerCase().trim(); + const ei = SENIORITY_INDEX[e]; + if (ei === undefined) { + return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: `unknown seniority "${entity}"` }; + } + let best = 0; + let bestTarget = ""; + for (const t of targets) { + const ti = SENIORITY_INDEX[t.toLowerCase().trim()]; + if (ti === undefined) + continue; + const d = Math.abs(ei - ti); + let v = 0; + if (d === 0) + v = 1.0; + else if (d === 1) + v = 0.6; + else if (d === 2) + v = 0.2; + if (v > best) { + best = v; + bestTarget = t; + } + } + return { + value: best, weight: DEFAULT_WEIGHTS.seniority, + reason: best === 1 ? `exact seniority match (${entity})` + : best > 0 ? `seniority ${entity} ≈ target ${bestTarget}` + : `seniority ${entity} too far from targets`, + }; +} +function stem(token) { + let t = token.toLowerCase().replace(/[^a-z]/g, ""); + if (t.endsWith("ing") && t.length > 5) + t = t.slice(0, -3); + else if (t.endsWith("ed") && t.length > 4) + t = t.slice(0, -2); + else if (t.endsWith("es") && t.length > 4) + t = t.slice(0, -2); + else if (t.endsWith("s") && t.length > 3) + t = t.slice(0, -1); + return t; +} +function tokenSet(s) { + if (!s) + return new Set(); + return new Set(s.split(/[\s,/&-]+/).map(stem).filter((t) => t.length > 1)); +} +export function scoreFunction(entityDept, targets) { + if (!entityDept || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.function, reason: "no function/dept data" }; + } + const ent = tokenSet(entityDept); + let best = 0; + let bestTarget = ""; + for (const t of targets) { + const tgt = tokenSet(t); + if (!tgt.size) + continue; + let hits = 0; + for (const tok of tgt) + if (ent.has(tok)) + hits++; + const jacc = hits / Math.max(1, new Set([...ent, ...tgt]).size); + if (jacc > best) { + best = jacc; + bestTarget = t; + } + } + return { + value: Math.min(1, best * 1.5), + weight: DEFAULT_WEIGHTS.function, + reason: best > 0 ? `function "${entityDept}" overlaps "${bestTarget}"` : `function "${entityDept}" no overlap`, + }; +} +export function scoreIndustry(entityIndustries, targets) { + if (!entityIndustries.length || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.industry, reason: "no industry data" }; + } + const tset = new Set(targets.map((t) => t.toLowerCase())); + let best = 0; + let bestNote = ""; + for (const ei of entityIndustries) { + const e = ei.toLowerCase(); + if (tset.has(e)) { + best = 1.0; + bestNote = `industry "${ei}" matches target`; + break; + } + const parents = INDUSTRY_PARENTS[e] ?? []; + for (const p of parents) { + if (tset.has(p)) { + if (best < 0.7) { + best = 0.7; + bestNote = `industry "${ei}" ⊂ target "${p}"`; + } + } + } + } + if (!bestNote) + bestNote = `industries [${entityIndustries.join(",")}] don't match targets`; + return { value: best, weight: DEFAULT_WEIGHTS.industry, reason: bestNote }; +} +export function scoreCompanySize(emp, minE, maxE) { + if (emp == null || (minE == null && maxE == null)) { + return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: "company size unknown" }; + } + const lo = minE ?? 0; + const hi = maxE ?? Number.POSITIVE_INFINITY; + if (emp >= lo && emp <= hi) { + return { value: 1.0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} in target [${lo}, ${maxE ?? "∞"}]` }; + } + const dLo = emp < lo ? (lo - emp) / Math.max(1, lo) : 0; + const dHi = emp > hi ? (emp - hi) / Math.max(1, hi) : 0; + const d = Math.max(dLo, dHi); + if (d <= 0.5) + return { value: 0.5, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} adjacent to target` }; + return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} outside target [${lo}, ${maxE ?? "∞"}]` }; +} +export function scoreStage(entityStages, targets) { + if (!entityStages.length || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.stage, reason: "stage unknown" }; + } + let best = 0; + let note = ""; + for (const e of entityStages) { + const ei = STAGE_INDEX[e.toLowerCase().replace(/[-\s]/g, "_")]; + if (ei === undefined) + continue; + for (const t of targets) { + const ti = STAGE_INDEX[t.toLowerCase().replace(/[-\s]/g, "_")]; + if (ti === undefined) + continue; + const d = Math.abs(ei - ti); + let v = 0; + if (d === 0) + v = 1.0; + else if (d === 1) + v = 0.6; + if (v > best) { + best = v; + note = d === 0 ? `stage ${e} matches target` : `stage ${e} adjacent to ${t}`; + } + } + } + return { value: best, weight: DEFAULT_WEIGHTS.stage, reason: note || `stages [${entityStages.join(",")}] don't match targets` }; +} +// --------------------------------------------------------------------------- +// Geo: Haversine + exponential decay when coordinates are present; ISO2 +// + continent fallback otherwise. +// --------------------------------------------------------------------------- +const EARTH_KM = 6371; +export function haversineKm(lat1, lng1, lat2, lng2) { + const toRad = (x) => (x * Math.PI) / 180; + const dLat = toRad(lat2 - lat1); + const dLng = toRad(lng2 - lng1); + const a = Math.sin(dLat / 2) ** 2 + + Math.cos(toRad(lat1)) * Math.cos(toRad(lat2)) * Math.sin(dLng / 2) ** 2; + return 2 * EARTH_KM * Math.asin(Math.min(1, Math.sqrt(a))); +} +export function scoreGeo(input) { + const { entityIso2, targets } = input; + const hasCenter = input.centerLat != null && input.centerLng != null && (input.radiusKm ?? 0) > 0; + // Coordinate path: Haversine + exp(-d/radius) decay. + let entLat = input.entityLat ?? null; + let entLng = input.entityLng ?? null; + if ((entLat == null || entLng == null) && entityIso2) { + const c = COUNTRY_CENTROID[entityIso2.toLowerCase()]; + if (c) { + entLat = c[0]; + entLng = c[1]; + } + } + if (hasCenter && entLat != null && entLng != null) { + const d = haversineKm(input.centerLat, input.centerLng, entLat, entLng); + const v = Math.max(0, Math.min(1, Math.exp(-d / Math.max(1, input.radiusKm)))); + return { + value: v, + weight: DEFAULT_WEIGHTS.geo, + reason: `geo ${d.toFixed(0)}km from persona center (radius ${input.radiusKm}km) ⇒ ${v.toFixed(2)}`, + data: { distance_km: d, radius_km: input.radiusKm }, + }; + } + // ISO2 fallback. + if (!entityIso2 || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: "geo unknown" }; + } + const e = entityIso2.toLowerCase(); + const t = targets.map((x) => x.toLowerCase()); + if (t.includes(e)) { + return { value: 1.0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} matches target` }; + } + const ec = CONTINENT[e]; + if (ec) { + for (const tc of t) + if (CONTINENT[tc] === ec) { + return { value: 0.5, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} shares region with target ${tc.toUpperCase()}` }; + } + } + return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} not in targets [${targets.join(",")}]` }; +} +// --------------------------------------------------------------------------- +// Aggregation + rationale. +// --------------------------------------------------------------------------- +export function aggregate(components) { + let sum = 0; + let wsum = 0; + for (const k of Object.keys(components)) { + const c = components[k]; + sum += c.value * c.weight; + wsum += c.weight; + } + return wsum > 0 ? sum / wsum : 0; +} +export function buildRationale(personaName, entityName, employerName, components, score) { + const pct = Math.round(score * 100); + const top = Object.entries(components) + .map(([k, c]) => ({ k, contribution: c.value * c.weight, reason: c.reason })) + .sort((a, b) => b.contribution - a.contribution) + .slice(0, 3) + .map((x) => x.reason) + .join("; "); + const who = entityName ?? "entity"; + const where = employerName ? ` at ${employerName}` : ""; + return `${who}${where} scores ${pct}% against persona "${personaName}". Top drivers: ${top}.`; +} +function arrFromJson(s) { + if (!s) + return []; + try { + const v = JSON.parse(s); + return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; + } + catch { + return []; + } +} +function objFromJson(s) { + if (!s) + return {}; + try { + const v = JSON.parse(s); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +export function extractTargets(row) { + const hard = objFromJson(row.hard_filters_json); + const stagesFromHard = Array.isArray(hard.stages) ? hard.stages.filter((x) => typeof x === "string") + : Array.isArray(hard.target_stage) ? hard.target_stage.filter((x) => typeof x === "string") + : []; + const center = hard.geo_center && typeof hard.geo_center === "object" ? hard.geo_center : {}; + const titles = arrFromJson(row.buyer_titles_json); + const seniority = arrFromJson(row.buyer_seniority_json); + const functions = arrFromJson(row.buyer_departments_json); + const industries = arrFromJson(row.industries_json); + const geos = arrFromJson(row.geos_json); + // Task #8 spec: title_sim must use ONLY structured persona target + // fields — no long-form notes (thesis, free text). Embedding here is + // restricted to titles + seniority + function so the component score + // stays explainable and reproducible. + const title_text = [ + titles.join(", "), + seniority.length ? `Seniority: ${seniority.join(", ")}` : "", + functions.length ? `Function: ${functions.join(", ")}` : "", + ].filter(Boolean).join(". "); + return { + title_text, + titles, + seniority, + functions, + industries, + size_min: row.size_min ?? null, + size_max: row.size_max ?? null, + stages: stagesFromHard, + geos, + geo_center_lat: typeof center.lat === "number" ? center.lat : null, + geo_center_lng: typeof center.lng === "number" ? center.lng : null, + geo_radius_km: typeof center.radius_km === "number" ? center.radius_km + : typeof hard.radius_km === "number" ? hard.radius_km : null, + }; +} diff --git a/apps/worker/test-dist-q/services/personas/kinds/_generic.js b/apps/worker/test-dist-q/services/personas/kinds/_generic.js new file mode 100644 index 00000000..7af6e0ef --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/_generic.js @@ -0,0 +1,45 @@ +// Task #3: Generic kind plugin used as the default for kinds that +// don't ship a bespoke matcher. Drives candidate selection from the +// taxonomy's `targets` (entity kind) + `roles` (entity_roles.role IN) +// and delegates scoring to the existing PersonaMatchingService scorer +// for person targets. Company/fund targets currently fall back to the +// legacy persona_matches/accounts/buyers code path via the dispatcher. +import { loadPersonEntity, scoreEntityForPersona as scorePersonForPersona } from "../../personaMatching"; +import { KINDS } from "./taxonomy"; +export function makeGenericPlugin(kind) { + const def = KINDS[kind]; + return { + kind, + defaultEntityFilter(_persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + const binds = [def.targets]; + let sql = `SELECT DISTINCT e.id FROM u_entities e`; + if (def.roles.length) { + sql += ` JOIN entity_roles r ON r.entity_id = e.id`; + } + sql += ` WHERE e.kind = ? AND e.status = 'active'`; + if (def.roles.length) { + sql += ` AND r.role IN (${def.roles.map(() => "?").join(",")})`; + binds.push(...def.roles); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(env, persona, entityId) { + // Default behavior: only person targets are scored via the + // graph scorer. Fund/company targets are out of scope for the + // person-graph matcher and return null so the caller skips them. + if (def.targets !== "person") + return null; + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scorePersonForPersona(env, persona, entity); + }, + explainMatch(entityId) { + return `kind=${kind} target=${def.targets}${def.roles.length ? " roles=" + def.roles.join("|") : ""} entity=${entityId}`; + }, + }; +} diff --git a/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js b/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js new file mode 100644 index 00000000..a56a2ac2 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=academic_researcher. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AcademicResearcherPlugin = makeGenericPlugin("academic_researcher"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/account_company.js b/apps/worker/test-dist-q/services/personas/kinds/account_company.js new file mode 100644 index 00000000..8072a299 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/account_company.js @@ -0,0 +1,17 @@ +// Task #3: account_company kind plugin (legacy "account" kind). +// +// Sales-side accounts live in the legacy `accounts` table and are +// scored via personas/score.ts + persona_matches (Task #46), not via +// the u_entities person graph. This plugin exists so the dispatcher +// can identify the kind, but defaultEntityFilter returns an empty +// candidate set — the legacy code path in routes/personas.ts owns +// account rescoring end-to-end. +export const accountCompanyPlugin = { + kind: "account_company", + defaultEntityFilter(_persona, _opts) { + // Legacy path: accounts are not in u_entities for persona matching. + return { sql: `SELECT id FROM u_entities WHERE 1 = 0`, binds: [] }; + }, + async scoreEntity(_env, _persona, _entityId) { return null; }, + explainMatch(entityId) { return `account_company (legacy accounts table): entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/acquirer.js b/apps/worker/test-dist-q/services/personas/kinds/acquirer.js new file mode 100644 index 00000000..72f5b2c9 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/acquirer.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=acquirer. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AcquirerPlugin = makeGenericPlugin("acquirer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js b/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js new file mode 100644 index 00000000..d4d9efde --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=angel_individual. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AngelIndividualPlugin = makeGenericPlugin("angel_individual"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js b/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js new file mode 100644 index 00000000..40cbde98 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=beta_tester. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const BetaTesterPlugin = makeGenericPlugin("beta_tester"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js b/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js new file mode 100644 index 00000000..6f8f008e --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js @@ -0,0 +1,37 @@ +// Task #3: buyer_person kind plugin (legacy "buyer" kind). +// +// Combines: (a) the new u_entities person-graph filter on role IN +// ('buyer','decision_maker','champion'), (b) delegation to the +// person-graph scorer. The legacy `buyers` table is still rescored +// by personas/rescore.ts under the hood; this plugin only governs +// the new u_entities-backed candidate list. +import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; +const ROLES = ["buyer", "decision_maker", "champion"]; +export const buyerPersonPlugin = { + kind: "buyer_person", + defaultEntityFilter(_persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + const ph = ROLES.map(() => "?").join(","); + return { + // No role filter when entity_roles lacks buyer-flavored rows yet + // — fall back to all active person entities. Matches the legacy + // behavior so existing 'buyer' personas don't regress to empty. + sql: `SELECT e.id FROM u_entities e + WHERE e.kind = 'person' AND e.status = 'active' + AND ( + EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id AND r.role IN (${ph})) + OR NOT EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id) + ) + ORDER BY e.id LIMIT ? OFFSET ?`, + binds: [...ROLES, limit, offset], + }; + }, + async scoreEntity(env, persona, entityId) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scoreEntityForPersona(env, persona, entity); + }, + explainMatch(entityId) { return `buyer_person match: entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js b/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js new file mode 100644 index 00000000..3eb5cc89 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=channel_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ChannelPartnerPlugin = makeGenericPlugin("channel_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js b/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js new file mode 100644 index 00000000..c3786ce0 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=co_founder_match. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const CoFounderMatchPlugin = makeGenericPlugin("co_founder_match"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/competitor.js b/apps/worker/test-dist-q/services/personas/kinds/competitor.js new file mode 100644 index 00000000..3801e793 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/competitor.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=competitor. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const CompetitorPlugin = makeGenericPlugin("competitor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/design_partner.js b/apps/worker/test-dist-q/services/personas/kinds/design_partner.js new file mode 100644 index 00000000..04200099 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/design_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=design_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const DesignPartnerPlugin = makeGenericPlugin("design_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js b/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js new file mode 100644 index 00000000..ecebae4a --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=engineering_hire. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const EngineeringHirePlugin = makeGenericPlugin("engineering_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js b/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js new file mode 100644 index 00000000..a45b4674 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=executive_hire. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ExecutiveHirePlugin = makeGenericPlugin("executive_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/founder.js b/apps/worker/test-dist-q/services/personas/kinds/founder.js new file mode 100644 index 00000000..6ef92773 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/founder.js @@ -0,0 +1,68 @@ +// Task #3: founder kind plugin. +// +// Surfaces hint fields `founded_count`, `prior_exits`, `domain_expertise`. +// Match: entities with role IN ('founder','ceo','co_founder'). Hint +// numerics are validated by the form; we re-validate here so a stale +// or hand-edited persona can't crash the scorer. +import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; +const ROLES = ["founder", "ceo", "co_founder"]; +function readHint(persona, field) { + if (!persona.hard_filters_json) + return null; + try { + const j = JSON.parse(persona.hard_filters_json); + const v = j?.hints?.[field]; + return v == null ? null : String(v); + } + catch { + return null; + } +} +function splitCsv(v) { + return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; +} +export const founderPlugin = { + kind: "founder", + defaultEntityFilter(persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + // Hints become hard predicates so they actually narrow the + // candidate set (not just decorate the UI). Roles in entity_roles + // act as a tag space — domain expertise becomes a 'domain:' + // tag, prior_exits / founded_count map to 'exits:N+' / 'founded:N+' + // synthetic tags that the enrichment pipeline emits per founder. + const domains = splitCsv(readHint(persona, "domain_expertise")); + const foundedMin = parseInt(readHint(persona, "founded_count") ?? "", 10); + const exitsMin = parseInt(readHint(persona, "prior_exits") ?? "", 10); + const binds = []; + const rolePh = ROLES.map(() => "?").join(","); + let sql = `SELECT DISTINCT e.id FROM u_entities e + JOIN entity_roles r ON r.entity_id = e.id + WHERE e.kind = 'person' AND e.status = 'active' + AND r.role IN (${rolePh})`; + binds.push(...ROLES); + if (domains.length) { + const ph = domains.map(() => "?").join(","); + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rd WHERE rd.entity_id = e.id AND rd.role IN (${ph}))`; + binds.push(...domains.map((d) => `domain:${d}`)); + } + if (Number.isFinite(foundedMin) && foundedMin > 0) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rf WHERE rf.entity_id = e.id AND rf.role = ?)`; + binds.push(`founded:${foundedMin}+`); + } + if (Number.isFinite(exitsMin) && exitsMin > 0) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles re WHERE re.entity_id = e.id AND re.role = ?)`; + binds.push(`exits:${exitsMin}+`); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(env, persona, entityId) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scoreEntityForPersona(env, persona, entity); + }, + explainMatch(entityId) { return `founder match: entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js b/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js new file mode 100644 index 00000000..6bcd934d --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=fractional_executive. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const FractionalExecutivePlugin = makeGenericPlugin("fractional_executive"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js b/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js new file mode 100644 index 00000000..0c548ec6 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=government_grant_officer. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const GovernmentGrantOfficerPlugin = makeGenericPlugin("government_grant_officer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/index.js b/apps/worker/test-dist-q/services/personas/kinds/index.js new file mode 100644 index 00000000..ed940044 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/index.js @@ -0,0 +1,71 @@ +// Task #3: persona-kind plugin registry + dispatcher entrypoint. +// +// The PersonaMatchingService reads the persona's kind, calls +// getPluginFor(kind), and delegates to defaultEntityFilter / scoreEntity. +// Kinds without a bespoke plugin file fall back to the generic plugin +// driven by the taxonomy's `roles` array (see _generic.ts). +import { ALL_KIND_KEYS, KINDS, resolveKind } from "./taxonomy"; +import { investorPersonPlugin } from "./investor_person"; +import { investorFirmPlugin } from "./investor_firm"; +import { venturePartnerPlugin } from "./venture_partner"; +import { founderPlugin } from "./founder"; +import { accountCompanyPlugin } from "./account_company"; +import { buyerPersonPlugin } from "./buyer_person"; +import { AngelIndividualPlugin } from "./angel_individual"; +import { LimitedPartnerPlugin } from "./limited_partner"; +import { CoFounderMatchPlugin } from "./co_founder_match"; +import { ExecutiveHirePlugin } from "./executive_hire"; +import { EngineeringHirePlugin } from "./engineering_hire"; +import { FractionalExecutivePlugin } from "./fractional_executive"; +import { ChannelPartnerPlugin } from "./channel_partner"; +import { IntegrationPartnerPlugin } from "./integration_partner"; +import { DesignPartnerPlugin } from "./design_partner"; +import { BetaTesterPlugin } from "./beta_tester"; +import { JournalistAnalystPlugin } from "./journalist_analyst"; +import { ThoughtLeaderPlugin } from "./thought_leader"; +import { AcademicResearcherPlugin } from "./academic_researcher"; +import { GovernmentGrantOfficerPlugin } from "./government_grant_officer"; +import { RegulatorPlugin } from "./regulator"; +import { PolicyAdvisorPlugin } from "./policy_advisor"; +import { ServiceProviderPlugin } from "./service_provider"; +import { AcquirerPlugin } from "./acquirer"; +import { CompetitorPlugin } from "./competitor"; +const REGISTRY_MAP = { + account_company: accountCompanyPlugin, + buyer_person: buyerPersonPlugin, + investor_person: investorPersonPlugin, + investor_firm: investorFirmPlugin, + venture_partner: venturePartnerPlugin, + founder: founderPlugin, + angel_individual: AngelIndividualPlugin, + limited_partner: LimitedPartnerPlugin, + co_founder_match: CoFounderMatchPlugin, + executive_hire: ExecutiveHirePlugin, + engineering_hire: EngineeringHirePlugin, + fractional_executive: FractionalExecutivePlugin, + channel_partner: ChannelPartnerPlugin, + integration_partner: IntegrationPartnerPlugin, + design_partner: DesignPartnerPlugin, + beta_tester: BetaTesterPlugin, + journalist_analyst: JournalistAnalystPlugin, + thought_leader: ThoughtLeaderPlugin, + academic_researcher: AcademicResearcherPlugin, + government_grant_officer: GovernmentGrantOfficerPlugin, + regulator: RegulatorPlugin, + policy_advisor: PolicyAdvisorPlugin, + service_provider: ServiceProviderPlugin, + acquirer: AcquirerPlugin, + competitor: CompetitorPlugin, +}; +// Sanity check: every declared kind has a plugin file. If a future +// kind is added to taxonomy without a wrapper, surface it at boot. +for (const k of ALL_KIND_KEYS) { + if (!REGISTRY_MAP[k]) + throw new Error(`missing plugin file for kind=${k}`); +} +export function getPluginFor(rawKind) { + const k = resolveKind(rawKind) ?? "account_company"; + return REGISTRY_MAP[k]; +} +export { KINDS, ALL_KIND_KEYS, resolveKind }; +export * from "./taxonomy"; diff --git a/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js b/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js new file mode 100644 index 00000000..194ec5ec --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=integration_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const IntegrationPartnerPlugin = makeGenericPlugin("integration_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js b/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js new file mode 100644 index 00000000..c4af4a4f --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js @@ -0,0 +1,64 @@ +// Task #3: investor_firm kind plugin. +// +// Targets entities where kind='fund' (or 'firm' legacy) AND +// entity_roles.role='investor_firm'. The person-graph scorer doesn't +// apply to funds directly; matching here is structural (role + kind +// + AUM/stage hints) and we expose a deterministic surface so the +// dispatcher can present candidates even without per-entity scoring. +// Read a hint value from hard_filters_json.hints.. +function readHint(persona, field) { + if (!persona.hard_filters_json) + return null; + try { + const j = JSON.parse(persona.hard_filters_json); + const v = j?.hints?.[field]; + return typeof v === "string" && v ? v : null; + } + catch { + return null; + } +} +// Split a comma-separated hint value into trimmed tokens. +function splitCsv(v) { + return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; +} +export const investorFirmPlugin = { + kind: "investor_firm", + defaultEntityFilter(persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + // Roles in entity_roles act as a tag space. When the persona + // specifies aum_band or stage_focus hints, we require that the + // fund carries the corresponding tag-role (e.g. 'aum:$1B-$5B' or + // 'stage:seed'). This way the hints actually narrow the candidate + // set rather than just decorating the UI. + const aum = readHint(persona, "aum_band"); // e.g. "$1B-$5B" + const stages = splitCsv(readHint(persona, "stage_focus")); // e.g. ["seed","series_a"] + const binds = []; + let sql = `SELECT DISTINCT e.id FROM u_entities e + JOIN entity_roles r ON r.entity_id = e.id + WHERE e.status = 'active' + AND e.kind IN ('fund','firm') + AND r.role = 'investor_firm'`; + if (aum) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles ra WHERE ra.entity_id = e.id AND ra.role = ?)`; + binds.push(`aum:${aum}`); + } + if (stages.length) { + const ph = stages.map(() => "?").join(","); + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rs WHERE rs.entity_id = e.id AND rs.role IN (${ph}))`; + binds.push(...stages.map((s) => `stage:${s}`)); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(_env, _persona, _entityId) { + // Structural-only match for funds — no per-entity scoring at this + // tier. Dispatcher treats null as "candidate present, score 0.5". + return null; + }, + explainMatch(entityId) { + return `investor_firm structural match: entity=${entityId} role=investor_firm kind=fund|firm`; + }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_person.js b/apps/worker/test-dist-q/services/personas/kinds/investor_person.js new file mode 100644 index 00000000..dfec68e4 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/investor_person.js @@ -0,0 +1,14 @@ +// Task #3: investor_person kind plugin. +// +// Acceptance criteria: matches entities where type='person' AND +// entity_roles.role IN ('investor','vc','gp','partner_at_firm'), +// then filters by firm size / stage / sector criteria via the +// existing person-graph scorer (which already considers employer +// sectors / stages / employees from career_history + entity_summary). +import { makeGenericPlugin } from "./_generic"; +// The generic plugin's defaultEntityFilter already picks up the +// taxonomy's role list ['investor','vc','gp','partner_at_firm'] and +// the person scorer already weighs employer sector / stage / size. +// We export it under a stable name so the registry can swap in a +// bespoke implementation later without touching the registry wiring. +export const investorPersonPlugin = makeGenericPlugin("investor_person"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js b/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js new file mode 100644 index 00000000..086a2293 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=journalist_analyst. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const JournalistAnalystPlugin = makeGenericPlugin("journalist_analyst"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js b/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js new file mode 100644 index 00000000..ace8b1a2 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=limited_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const LimitedPartnerPlugin = makeGenericPlugin("limited_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js b/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js new file mode 100644 index 00000000..39706ce5 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=policy_advisor. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const PolicyAdvisorPlugin = makeGenericPlugin("policy_advisor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/regulator.js b/apps/worker/test-dist-q/services/personas/kinds/regulator.js new file mode 100644 index 00000000..0219b79f --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/regulator.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=regulator. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const RegulatorPlugin = makeGenericPlugin("regulator"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/service_provider.js b/apps/worker/test-dist-q/services/personas/kinds/service_provider.js new file mode 100644 index 00000000..4abb6751 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/service_provider.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=service_provider. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ServiceProviderPlugin = makeGenericPlugin("service_provider"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js b/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js new file mode 100644 index 00000000..981f0019 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js @@ -0,0 +1,86 @@ +// Task #3: Expand persona kinds taxonomy. +// +// Single source of truth for the persona-kind taxonomy. Both the form +// (consumed via GET /api/personas/taxonomy) and the matcher dispatcher +// read from this module. No taxonomy duplication in HTML or in plugin +// code — plugins reference KINDS[kind] to discover their group, label, +// allowed criteria sections, and required hint fields. +// Hint-field metadata for the form (label, type, options). +export const HINTS = { + subtype: { label: "Subtype", type: "select", options: ["lawyer", "banker", "operator", "politician", "scout", "advisor", "board_member"] }, + aum_band: { label: "AUM band", type: "select", options: ["<$50M", "$50M-$250M", "$250M-$1B", "$1B-$5B", ">$5B"] }, + stage_focus: { label: "Stage focus", type: "text", placeholder: "pre_seed, seed, series_a" }, + founded_count: { label: "Companies founded (min)", type: "number" }, + prior_exits: { label: "Prior exits (min)", type: "number" }, + domain_expertise: { label: "Domain expertise", type: "text", placeholder: "fintech, dev_tools" }, +}; +const COMMON = ["geography", "industry", "signals", "tuning"]; +const PERSON_BASE = [...COMMON, "buyer_profile"]; +const COMPANY_BASE = ["sizing", ...COMMON, "tech_stack"]; +export const KINDS_LIST = [ + // ---- Sales + { kind: "account_company", group: "Sales", label: "Account (company)", sections: COMPANY_BASE, hints: [], targets: "company", roles: [] }, + { kind: "buyer_person", group: "Sales", label: "Buyer (person)", sections: PERSON_BASE, hints: [], targets: "person", roles: ["buyer", "decision_maker", "champion"] }, + // ---- Capital + { kind: "investor_firm", group: "Capital", label: "Investor firm", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band", "stage_focus"], targets: "fund", roles: ["investor_firm"] }, + { kind: "investor_person", group: "Capital", label: "Investor (person)", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["stage_focus"], targets: "person", roles: ["investor", "vc", "gp", "partner_at_firm"] }, + { kind: "angel_individual", group: "Capital", label: "Angel investor", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["angel", "investor"] }, + { kind: "limited_partner", group: "Capital", label: "Limited partner", sections: ["geography", "signals", "tuning"], hints: ["aum_band"], targets: "person", roles: ["limited_partner", "lp"] }, + { kind: "venture_partner", group: "Capital", label: "Venture partner", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["subtype", "domain_expertise"], targets: "person", roles: ["venture_partner", "advisor", "scout"] }, + // ---- People + { kind: "founder", group: "People", label: "Founder", sections: ["geography", "industry", "signals", "tuning"], hints: ["founded_count", "prior_exits", "domain_expertise"], targets: "person", roles: ["founder", "ceo"] }, + { kind: "co_founder_match", group: "People", label: "Co-founder match", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["founder", "engineer", "designer"] }, + { kind: "executive_hire", group: "People", label: "Executive hire", sections: PERSON_BASE, hints: ["domain_expertise"], targets: "person", roles: ["executive", "vp", "c_suite"] }, + { kind: "engineering_hire", group: "People", label: "Engineering hire", sections: ["geography", "tech_stack", "signals", "buyer_profile", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["engineer", "ic"] }, + { kind: "fractional_executive", group: "People", label: "Fractional executive", sections: PERSON_BASE, hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["fractional", "advisor", "executive"] }, + // ---- Partnerships + { kind: "channel_partner", group: "Partnerships", label: "Channel partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["partner", "reseller"] }, + { kind: "integration_partner", group: "Partnerships", label: "Integration partner", sections: ["sizing", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["partner", "integration"] }, + { kind: "design_partner", group: "Partnerships", label: "Design partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["customer", "prospect"] }, + { kind: "beta_tester", group: "Partnerships", label: "Beta tester", sections: ["geography", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["customer", "prospect", "beta"] }, + // ---- Influence (no tech_stack — content/coverage focused) + { kind: "journalist_analyst", group: "Influence", label: "Journalist / analyst", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["journalist", "analyst", "press"] }, + { kind: "thought_leader", group: "Influence", label: "Thought leader", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["influencer", "thought_leader", "speaker"] }, + { kind: "academic_researcher", group: "Influence", label: "Academic researcher", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["researcher", "academic", "professor"] }, + // ---- Public Sector + { kind: "government_grant_officer", group: "Public Sector", label: "Government grant officer", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["government", "grant_officer", "program_officer"] }, + { kind: "regulator", group: "Public Sector", label: "Regulator", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["regulator", "agency_official"] }, + { kind: "policy_advisor", group: "Public Sector", label: "Policy advisor", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["policy_advisor", "staffer", "aide"] }, + // ---- Operational + { kind: "service_provider", group: "Operational", label: "Service provider", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "company", roles: ["vendor", "service_provider", "agency"] }, + { kind: "acquirer", group: "Operational", label: "Acquirer", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band"], targets: "company", roles: ["acquirer", "strategic"] }, + { kind: "competitor", group: "Operational", label: "Competitor", sections: ["sizing", "geography", "industry", "tech_stack", "tuning"], hints: [], targets: "company", roles: ["competitor"] }, +]; +export const KINDS = Object.fromEntries(KINDS_LIST.map((k) => [k.kind, k])); +export const ALL_KIND_KEYS = KINDS_LIST.map((k) => k.kind); +// Legacy values from before Task #3 — map to the closest new kind so +// existing personas keep working without a backfill migration. +export const LEGACY_KIND_MAP = { + account: "account_company", + buyer: "buyer_person", +}; +export function resolveKind(raw) { + if (!raw) + return null; + if (KINDS[raw]) + return raw; + if (LEGACY_KIND_MAP[raw]) + return LEGACY_KIND_MAP[raw]; + return null; +} +export function isValidKind(raw) { + return resolveKind(raw) !== null; +} +// Grouped view for the form's grouped