diff --git a/apps/site/assets/js/bulk-bar.js b/apps/site/assets/js/bulk-bar.js index 78533a32..429cf5ab 100644 --- a/apps/site/assets/js/bulk-bar.js +++ b/apps/site/assets/js/bulk-bar.js @@ -130,7 +130,6 @@ var selection = new Map(); // id -> true var allMatchingMode = false; // true after 2nd header click var currentSignature = cfg.getFilterSignature ? cfg.getFilterSignature() : ""; - var headerCheck = document.getElementById("ads-bulk-header-check"); // Rehydrate selection on init iff the persisted signature matches // the current filter signature. Drop otherwise. @@ -211,15 +210,28 @@ } } - if (headerCheck) { - var headerClicks = 0; - headerCheck.addEventListener("click", function () { - headerClicks += 1; - if (headerClicks === 1) { selectPage(headerCheck.checked); } - else { headerClicks = 0; selectAllMatching(); } - setTimeout(function () { headerClicks = 0; }, 2500); - }); - } + // Delegated, not bound to the element. On the investors and companies + // pages the header checkbox is emitted inside the async row render, so it + // does not exist when init() runs and it is replaced wholesale on every + // non-append load. A direct listener resolved at init() caught neither + // case: "select page" and "select all matching" were dead on both pages, + // and the checkbox still ticked, so it looked like it had worked. + // (It did work on leads and accounts, whose header cell is static HTML — + // which is why the two pages behaved differently for the same code.) + var headerClicks = 0; + var headerResetTimer = null; + document.addEventListener("click", function (e) { + var hc = e.target && e.target.closest ? e.target.closest("#ads-bulk-header-check") : null; + if (!hc) return; + headerClicks += 1; + if (headerClicks === 1) { selectPage(hc.checked); } + else { headerClicks = 0; selectAllMatching(); } + // Cleared rather than stacked: the old code queued a fresh 2.5 s reset + // on every click, so an earlier timer could fire between the two clicks + // of a deliberate double-click and reset the counter mid-gesture. + if (headerResetTimer) clearTimeout(headerResetTimer); + headerResetTimer = setTimeout(function () { headerClicks = 0; }, 2500); + }); // Rebind row checks after each list refresh — pages call this manually // (or use a MutationObserver as a fallback). diff --git a/apps/site/assets/js/leads.js b/apps/site/assets/js/leads.js index 18633d06..ab3c419c 100644 --- a/apps/site/assets/js/leads.js +++ b/apps/site/assets/js/leads.js @@ -72,6 +72,23 @@ }).join(""); } + // Where a lead's name should go. + // + // This used to be an unconditional /dashboard/people/?id=. That page + // hands the value to /api/profilers/:entity_id/*, which is keyed on + // u_entities — and `l.id` is the legacy `leads` primary key, a different id + // space entirely. So the link never resolved: every name on the Leads page + // opened a person profile with every panel empty, which reads as "we have + // no data on this person" rather than "this link is wrong". + // + // The listing now carries `entity_id` when the lead has been mapped into + // u_entities. Rows that have not been mapped yet go to the lead detail + // page, which is keyed on exactly the id we have. + function nameHref(l) { + if (l.entity_id) return "/dashboard/people/?id=" + encodeURIComponent(l.entity_id); + return "/dashboard/lead/?id=" + encodeURIComponent(l.id); + } + function render(items) { var tbody = document.getElementById("ads-leads-tbody"); if (!items.length) { @@ -81,7 +98,7 @@ tbody.innerHTML = items.map(function (l) { return '' + '' - + '' + esc(l.name || "(no name)") + '' + rolesBadges(l.roles) + '' + + '' + esc(l.name || "(no name)") + '' + rolesBadges(l.roles) + '' + '' + esc(l.org || "—") + '' + '' + esc(l.email || "—") + '' + '' + esc(l.status || "—") + '' diff --git a/apps/site/assets/js/profile-tab.js b/apps/site/assets/js/profile-tab.js index 368def98..65acffb7 100644 --- a/apps/site/assets/js/profile-tab.js +++ b/apps/site/assets/js/profile-tab.js @@ -4,7 +4,8 @@ // window.ADS.Profile.mount({ rootId, entityId }) — embed mode // window.ADS.Profile.mountStandalone() — /dashboard/profile/ // -// Resolves entity via ?entity= or ?table=&ref= (same convention as DD/News). +// Resolves entity via ?entity= (or ?id=) or ?table=&ref= (same +// convention as DD/News). (function () { if (window.ADS && window.ADS.Profile) return; @@ -27,9 +28,19 @@ sec_rel: "Secular ↔ Religious", }; + // `?id=` is accepted alongside `?entity=` because seven navigation entry + // points across six pages link here with `?id=` — ops-garbage-review.js, + // predictions.html (twice), watchlists.html, power-nodes.html, + // dossiers.html and ops-quality.html. Reading only `entity` made every one + // of those land on "No entity selected." Fixing the reader rather than the + // seven call sites also covers any future link written the same way. function qs() { var p = new URLSearchParams(window.location.search); - return { entity: p.get("entity"), table: p.get("table"), ref: p.get("ref") }; + return { + entity: p.get("entity") || p.get("id"), + table: p.get("table"), + ref: p.get("ref"), + }; } function esc(s) { return String(s == null ? "" : s).replace(/[&<>"']/g, function (c) { @@ -399,7 +410,7 @@ if (!entityId) { if (titleEl) titleEl.textContent = "Profile"; var host = document.getElementById("ads-profile-root"); - if (host) host.innerHTML = '
No entity selected. Pass ?entity= or ?table=&ref=.
'; + if (host) host.innerHTML = '
No entity selected. Pass ?entity=, ?id= or ?table=&ref=.
'; return; } if (titleEl) titleEl.textContent = "Loading…"; diff --git a/apps/site/dashboard/crawlers.html b/apps/site/dashboard/crawlers.html index b9769305..35a18e90 100644 --- a/apps/site/dashboard/crawlers.html +++ b/apps/site/dashboard/crawlers.html @@ -41,10 +41,11 @@

Recent runs

Inserted Skipped Accts ↑ + Notes Error - Loading… + Loading… @@ -54,6 +55,20 @@

Recent runs

(async function () { const tbody = document.querySelector("#ads-crawlers-table tbody"); const runsBody = document.querySelector("#ads-crawler-runs tbody"); + // crawler_runs.meta_json holds the per-source counters. Rendering them is + // the difference between "this source had nothing to scan" and "this source + // scanned everything and found nothing new" — both of which otherwise show + // as a run with 0 emitted and status ok. The seven ATS sources are seeded + // from accounts.meta_json keys that nothing sets automatically, so + // seeded_accounts=0 is the answer to "why is this always empty?". + function runNotes(metaJson) { + if (!metaJson) return ""; + var m; + try { m = JSON.parse(metaJson); } catch (e) { return ""; } + if (!m || typeof m !== "object") return ""; + return Object.keys(m).map(function (k) { return k + "=" + m[k]; }).join(" · "); + } + function fmt(s) { return s ? new Date(s).toLocaleString() : "—"; } function esc(s) { return String(s == null ? "" : s).replace(/[&<>"']/g, function (c) { return ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" })[c]; }); } async function refresh() { @@ -87,11 +102,12 @@

Recent runs

${esc(x.signals_inserted)} ${esc(x.signals_skipped)} +${esc(x.accounts_created)} / ${esc(x.accounts_resolved)} + ${esc(runNotes(x.meta_json))} ${esc(x.error || "")} `).join(""); } catch (e) { - runsBody.innerHTML = `Failed to load: ${esc(e && e.message ? e.message : e)}`; + runsBody.innerHTML = `Failed to load: ${esc(e && e.message ? e.message : e)}`; } } document.addEventListener("change", async (e) => { diff --git a/apps/worker/package.json b/apps/worker/package.json index 30567bc8..633160e5 100644 --- a/apps/worker/package.json +++ b/apps/worker/package.json @@ -10,7 +10,7 @@ "lint": "eslint 'src/**/*.ts'", "lint:fix": "eslint 'src/**/*.ts' --fix", "gates": "bash scripts/gates.sh", - "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs", + "test": "tsc -p tsconfig.test.json && node --test test/profile.test.mjs test/agent.test.mjs test/monitoring.test.mjs test/osint.test.mjs test/profile-helpers.test.mjs test/profile-db.test.mjs test/csv_person_import.test.mjs test/firm_geo_backfill.test.mjs test/profilers.test.mjs test/access_guard.test.mjs test/markMap.security.test.mjs test/profileWorkflows.test.mjs test/vc_sources.test.mjs src/crawler/adapters/__tests__/*.test.mjs src/crawler/adapters/deals/__tests__/*.test.mjs src/services/deals/__tests__/*.test.mjs src/ai/__tests__/*.test.mjs src/services/capTable/__tests__/*.test.mjs src/services/valuation/__tests__/*.test.mjs src/services/documents/__tests__/*.test.mjs test/verification.test.mjs test/verification.verifiers.test.mjs test/preflight.test.mjs test/proxyPool.test.mjs test/fetcher_proxy.test.mjs test/sweeper.test.mjs src/services/termSheets/__tests__/*.test.mjs src/services/fundReturns/__tests__/*.test.mjs src/services/edgeQuality/__tests__/*.test.mjs src/services/intros/__tests__/*.test.mjs src/services/diligence/__tests__/*.test.mjs src/services/founderCrm/__tests__/*.test.mjs src/services/mlOps/__tests__/*.test.mjs test/people.test.mjs test/overrides.test.mjs test/overrides_integration.test.mjs test/relationships_inference.test.mjs src/services/compute/__tests__/*.test.mjs src/services/systemHealth/__tests__/*.test.mjs test/investor_portfolio.test.mjs test/error_surfacing.test.mjs test/subrequest_budget.test.mjs test/errors_classify.test.mjs test/robots_skip.test.mjs test/do_merge_consistency.test.mjs test/search_sync_index.test.mjs test/comp_panel_snapshot.test.mjs test/account_score_batch.test.mjs test/accounts_sort_fallback.test.mjs test/dd_scores_by_ref.test.mjs test/job_state_transitions.test.mjs test/simple_request.test.mjs test/simple_request_client.test.mjs test/pagination_guard.test.mjs test/enrichment_budget.test.mjs test/org_entity_merge.test.mjs test/import_entity_mapping.test.mjs test/schema_drift.test.mjs test/garbage.test.mjs test/crawler_seeds.test.mjs test/identity_harvest.test.mjs test/personaMatching.test.mjs test/personaMatchingAcceptance.test.mjs test/personaMatchingDbContract.test.mjs test/profile_comments.test.mjs test/profiler_batch.test.mjs test/schema_drift_repo.test.mjs test/migrations_apply.test.mjs test/merge_repoints.test.mjs test/ci_wrangler_version.test.mjs test/soft_delete_roundtrip.test.mjs test/site_deeplinks.test.mjs test/persona_headcount.test.mjs test/crawler_run_meta.test.mjs test/diligence_founder_link.test.mjs test/edge_quality_sectors.test.mjs test/sector_resolution.test.mjs test/team_snapshot_eligibility.test.mjs", "cf:provision": "node scripts/provision-cf.mjs" }, "dependencies": { diff --git a/apps/worker/src/entities/backfill.ts b/apps/worker/src/entities/backfill.ts index 8b160d3b..c4b38751 100644 --- a/apps/worker/src/entities/backfill.ts +++ b/apps/worker/src/entities/backfill.ts @@ -79,7 +79,7 @@ export async function backfillAccounts(env: Env, offset = 0, limit = BATCH): Pro `SELECT id, name, legal_name, website, domain, industry, industries_json, hq_country_iso2, hq_region, hq_city, funding_stage, linkedin_url, twitter_handle, github_org, crunchbase_url, - fit_score, intent_score + fit_score, intent_score, employees FROM accounts ORDER BY created_at LIMIT ? OFFSET ?`, ).bind(limit + 1, offset).all>(); const rows = r.results ?? []; diff --git a/apps/worker/src/entities/dualwrite.ts b/apps/worker/src/entities/dualwrite.ts index 83045e4f..32858121 100644 --- a/apps/worker/src/entities/dualwrite.ts +++ b/apps/worker/src/entities/dualwrite.ts @@ -100,6 +100,7 @@ interface AccountLikeInput { crunchbase_url?: string | null; fit_score?: number | null; intent_score?: number | null; + employees?: number | null; } interface BuyerLikeInput { @@ -408,6 +409,13 @@ export async function syncAccountToEntity(env: Env, a: AccountLikeInput, source { predicate: "funding_stage", value_text: a.funding_stage ?? null }, { predicate: "fit_max_score", value_number: numOrNull(a.fit_score) }, { predicate: "intent_score", value_number: numOrNull(a.intent_score) }, + // `accounts.employees` was the one sizing column dualwrite dropped, so + // no account entity carried a headcount fact and persona matching + // scored every one of them "company size unknown". `employees` is the + // predicate the registry declares (entities/profile-predicates.ts) and + // the one secEdgar/persist.ts already writes, so this converges on the + // name that exists rather than adding a fourth spelling. + { predicate: "employees", value_number: numOrNull(a.employees) }, ]; await insertFactsBatch(env, entityId, patches, source, "scrape"); await Promise.all([ diff --git a/apps/worker/src/entities/sector.ts b/apps/worker/src/entities/sector.ts new file mode 100644 index 00000000..db079eee --- /dev/null +++ b/apps/worker/src/entities/sector.ts @@ -0,0 +1,135 @@ +// One place that knows how a sector is actually stored. +// +// Four call sites in the valuation module and one in the edge-quality sweep +// each asked `facts` for `company.sector`, `firm.sector` or `sector`. No +// writer in the worker produces any of those. What is written is: +// +// * `firm.sectors` / `company.sectors` — a JSON ARRAY in value_json, +// emitted by the profile workflows (crawler/profileWorkflows/_commonSchemas). +// * `industry` — value_text, emitted by the account dual-write. +// * `entity_summary.sectors_csv` — the materialised, deduped, slugged list +// the summary rebuild derives from `sector` tags. +// +// Reading only the three singular text predicates meant every sector lookup +// returned nothing. In the comp panel that is worse than empty: the screen +// `continue`s past any candidate whose sector does not match, so an operator +// filtering by sector got an empty panel and the reasonable conclusion that +// there were no comparable companies. +// +// These helpers exist so the next reader does not have to rediscover which of +// the six spellings is the live one. + +import type { Env } from "../types"; + +/** Predicates whose value lives in value_text. */ +export const SECTOR_TEXT_PREDICATES = [ + "entity.primary_sector", "company.sector", "firm.sector", + "sector", "industry", "firm.industry", +] as const; + +/** Predicates whose value is a JSON array in value_json. */ +export const SECTOR_ARRAY_PREDICATES = [ + "firm.sectors", "company.sectors", "sectors", +] as const; + +/** + * True when `entityId` carries `sector` under any storage shape. + * + * Written as one statement so a per-row screen costs one round trip. The + * summary arm uses comma-delimited `instr` — the same technique + * monitoring/smart.ts uses against the same column — because `sectors_csv` + * is a bare join of slugs, so a substring test alone would match "fin" inside + * "fintech". + */ +export async function entityHasSector(env: Env, entityId: string, sector: string): Promise { + const wanted = sector.trim(); + if (!entityId || !wanted) return false; + const r = await env.DB.prepare( + `SELECT 1 AS hit FROM entity_summary s + WHERE s.entity_id = ? + AND s.sectors_csv IS NOT NULL + AND instr(',' || lower(s.sectors_csv) || ',', ',' || lower(?) || ',') > 0 + UNION ALL + SELECT 1 AS hit FROM facts f + WHERE f.entity_id = ? + AND f.is_current = 1 + AND ( + (f.predicate IN ('entity.primary_sector','company.sector','firm.sector','sector','industry','firm.industry') + AND lower(trim(f.value_text)) = lower(?)) + OR (f.predicate IN ('firm.sectors','company.sectors','sectors') + AND f.value_json IS NOT NULL + AND EXISTS (SELECT 1 FROM json_each(f.value_json) je + WHERE lower(trim(je.value)) = lower(?))) + ) + LIMIT 1`, + ).bind(entityId, wanted, entityId, wanted, wanted).first<{ hit: number }>(); + return Boolean(r); +} + +/** + * The entity's primary sector, lower-cased, or null when there is no evidence. + * Prefers `entity_summary` because it is the materialised list; falls back to + * facts for entities whose summary has not been rebuilt yet. + */ +export async function entityPrimarySector(env: Env, entityId: string): Promise { + if (!entityId) return null; + const sum = await env.DB.prepare( + `SELECT sectors_csv FROM entity_summary WHERE entity_id = ?`, + ).bind(entityId).first<{ sectors_csv: string | null }>(); + const fromSummary = (sum?.sectors_csv ?? "").split(",").map((x) => x.trim()).find(Boolean); + if (fromSummary) return fromSummary.toLowerCase(); + + const f = await env.DB.prepare( + `SELECT value_text, value_json FROM facts + WHERE entity_id = ? + AND is_current = 1 + AND predicate IN ('entity.primary_sector','company.sector','firm.sector','sector', + 'industry','firm.industry','firm.sectors','company.sectors','sectors') + ORDER BY observed_at DESC + LIMIT 1`, + ).bind(entityId).first<{ value_text: string | null; value_json: string | null }>(); + const text = f?.value_text?.trim(); + if (text) return text.toLowerCase(); + return firstOfJsonArray(f?.value_json ?? null); +} + +/** First non-empty string in a JSON array column, lower-cased; null otherwise. */ +export function firstOfJsonArray(raw: string | null): string | null { + if (!raw) return null; + try { + const parsed = JSON.parse(raw) as unknown; + if (!Array.isArray(parsed)) return null; + for (const x of parsed) { + if (typeof x === "string" && x.trim()) return x.trim().toLowerCase(); + } + } catch { /* not JSON — nothing to take */ } + return null; +} + +/** + * `EXISTS (...)` fragment matching a sector against a column already in scope. + * + * A literal, not a template: the repo's SQL gate forbids interpolating into a + * statement, and the only variable part here would have been the column name. + * Callers that need a different column inline their own copy rather than + * building one by concatenation. + * + * Binds, in order: sector, sector, sector. + */ +export const SECTOR_MATCHES_COMPANY_ENTITY_SQL = `( + EXISTS (SELECT 1 FROM entity_summary s + WHERE s.entity_id = vm.company_entity_id + AND s.sectors_csv IS NOT NULL + AND instr(',' || lower(s.sectors_csv) || ',', ',' || lower(?) || ',') > 0) + OR EXISTS (SELECT 1 FROM facts f + WHERE f.entity_id = vm.company_entity_id + AND f.is_current = 1 + AND ( + (f.predicate IN ('entity.primary_sector','company.sector','firm.sector','sector','industry','firm.industry') + AND lower(trim(f.value_text)) = lower(?)) + OR (f.predicate IN ('firm.sectors','company.sectors','sectors') + AND f.value_json IS NOT NULL + AND EXISTS (SELECT 1 FROM json_each(f.value_json) je + WHERE lower(trim(je.value)) = lower(?))) + )) +)`; diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index 87447e4e..167f2b80 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -127,6 +127,17 @@ api.use( allowHeaders: ["Content-Type", "Cf-Access-Jwt-Assertion", "Idempotency-Key"], }), ); +// The public allow-list covers the CHEAP liveness probe, and only that. +// `/deep` shares the same router, so mounting the router publicly also +// published a readiness sweep that runs 18 binding probes, three COUNT(*) +// scans of error_log, and a live outbound fetchPage() with an 8 s budget +// that walks every fetcher tier — including the metered Browser Rendering +// tier when the cheaper tiers escalate. Unauthenticated, that is a cost +// amplifier anyone can drive in a loop, and nothing about "health is +// public" was ever meant to grant it. Guard the deep sweep specifically; +// registering the middleware BEFORE the router is what makes it run first. +api.use("/health/deep", accessGuard); +api.use("/api/health/deep", accessGuard); api.route("/health", health); api.route("/api/health", health); api.route("/api/webhooks/campaigns", campaignsWebhook); diff --git a/apps/worker/src/prospects/runCrawl.ts b/apps/worker/src/prospects/runCrawl.ts index 24b785fc..f9a2ef3e 100644 --- a/apps/worker/src/prospects/runCrawl.ts +++ b/apps/worker/src/prospects/runCrawl.ts @@ -51,10 +51,17 @@ export async function runSource(env: Env, mod: SourceModule, opts?: { force?: bo let drafts: SignalEventDraft[] = []; let nextCursor: string | null | undefined = cursor; let crawlError: string | undefined; + // CrawlResult.meta is documented as "per-source counters appended to + // crawler_runs.meta_json" and the column has existed since migration 161, + // but nothing ever read the field — every source's counters were dropped on + // the floor. That is what left a source with no work to do and a source that + // scanned everything and found nothing new both recorded as `0 events, ok`. + let crawlMeta: Record | undefined; try { const r = await mod.crawl({ env, cursor, maxEvents: MAX_EVENTS_PER_RUN, accountId: opts?.accountId }); drafts = (r.events ?? []).slice(0, MAX_EVENTS_PER_RUN); nextCursor = r.cursor === undefined ? cursor : r.cursor; + crawlMeta = r.meta; } catch (e) { crawlError = (e as Error).message; } @@ -131,14 +138,21 @@ export async function runSource(env: Env, mod: SourceModule, opts?: { force?: bo if (nextCursor) await setCursor(env, mod.slug, nextCursor); const status: RunOutcome["status"] = crawlError ? "error" : (skipped > 0 && inserted === 0 ? "partial" : "ok"); - await finalize(env, runId, status, { events: drafts.length, inserted, skipped, created, resolved, cursor: nextCursor ?? null, error: crawlError }); + await finalize(env, runId, status, { events: drafts.length, inserted, skipped, created, resolved, cursor: nextCursor ?? null, error: crawlError, meta: crawlMeta }); return { runId, source: mod.slug, status, events_emitted: drafts.length, signals_inserted: inserted, signals_skipped: skipped, accounts_created: created, accounts_resolved: resolved, error: crawlError }; } -async function finalize(env: Env, runId: string, status: string, m: { events: number; inserted: number; skipped: number; created: number; resolved: number; cursor: string | null; error?: string }): Promise { +async function finalize(env: Env, runId: string, status: string, m: { events: number; inserted: number; skipped: number; created: number; resolved: number; cursor: string | null; error?: string; meta?: Record }): Promise { + // Serialised defensively: a source is free to put anything in `meta`, and a + // value that cannot be stringified (a cycle, a BigInt) must not be the + // reason a run never records its outcome at all. + let metaJson: string | null = null; + if (m.meta && Object.keys(m.meta).length) { + try { metaJson = JSON.stringify(m.meta); } catch { metaJson = null; } + } await env.DB.prepare( - `UPDATE crawler_runs SET finished_at = ?, status = ?, events_emitted = ?, signals_inserted = ?, signals_skipped = ?, accounts_created = ?, accounts_resolved = ?, cursor = ?, error = ? WHERE id = ?`, - ).bind(new Date().toISOString(), status, m.events, m.inserted, m.skipped, m.created, m.resolved, m.cursor, m.error ?? null, runId) + `UPDATE crawler_runs SET finished_at = ?, status = ?, events_emitted = ?, signals_inserted = ?, signals_skipped = ?, accounts_created = ?, accounts_resolved = ?, cursor = ?, error = ?, meta_json = ? WHERE id = ?`, + ).bind(new Date().toISOString(), status, m.events, m.inserted, m.skipped, m.created, m.resolved, m.cursor, m.error ?? null, metaJson, runId) .run().catch((e) => console.warn("crawler_runs finalize failed", e.message)); } diff --git a/apps/worker/src/prospects/sources/ashby.ts b/apps/worker/src/prospects/sources/ashby.ts index 1f2569e5..e929359c 100644 --- a/apps/worker/src/prospects/sources/ashby.ts +++ b/apps/worker/src/prospects/sources/ashby.ts @@ -23,13 +23,16 @@ const mod: SourceModule = { const rows = await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%ashby_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).ashby_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://jobs.ashbyhq.com/${encodeURIComponent(company)}.json`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: AshbyResp = {}; try { parsed = JSON.parse(res.body) as AshbyResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "ashby", res.body, "json"); @@ -50,7 +53,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `ashby_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/greenhouse.ts b/apps/worker/src/prospects/sources/greenhouse.ts index 0f836a10..a9e50d41 100644 --- a/apps/worker/src/prospects/sources/greenhouse.ts +++ b/apps/worker/src/prospects/sources/greenhouse.ts @@ -35,12 +35,15 @@ const mod: SourceModule = { LIMIT 100`, ).all(); let newest = since; + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let token = ""; try { token = String((JSON.parse(r.meta_json ?? "{}") as Record).greenhouse_board ?? ""); } catch { /* skip */ } if (!token) continue; + seeded += 1; const fetched = await fetchBoard(ctx.env, token); if (!fetched) continue; + boardsFetched += 1; const r2_key = await archiveRaw(ctx.env, "greenhouse", fetched.raw, "json"); const fresh = fetched.jobs.filter((j) => Date.parse(j.updated_at) > since); // Cluster: >= 5 new postings inside this run = hiring_burst. @@ -60,7 +63,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `greenhouse_board` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/lever.ts b/apps/worker/src/prospects/sources/lever.ts index 311f7ab9..0c0e8296 100644 --- a/apps/worker/src/prospects/sources/lever.ts +++ b/apps/worker/src/prospects/sources/lever.ts @@ -21,13 +21,16 @@ const mod: SourceModule = { const rows = await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%lever_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).lever_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://api.lever.co/v0/postings/${encodeURIComponent(company)}?mode=json`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let postings: Posting[] = []; try { postings = JSON.parse(res.body) as Posting[]; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "lever", res.body, "json"); @@ -47,7 +50,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? String(newest) : ctx.cursor }; + return { + events, + cursor: newest > since ? String(newest) : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `lever_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/personio.ts b/apps/worker/src/prospects/sources/personio.ts index 8d8e5dc2..776ec266 100644 --- a/apps/worker/src/prospects/sources/personio.ts +++ b/apps/worker/src/prospects/sources/personio.ts @@ -53,13 +53,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%personio_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).personio_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://${encodeURIComponent(company)}.jobs.personio.de/xml`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/xml" }); if (!res || !res.ok) continue; + boardsFetched += 1; const r2_key = await archiveRaw(ctx.env, "personio", res.body, "xml"); const positions = parsePositions(res.body); const fresh = positions.filter((p) => { @@ -82,7 +85,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `personio_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/recruitee.ts b/apps/worker/src/prospects/sources/recruitee.ts index 60fd329f..df40dab8 100644 --- a/apps/worker/src/prospects/sources/recruitee.ts +++ b/apps/worker/src/prospects/sources/recruitee.ts @@ -38,13 +38,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%recruitee_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let company = ""; try { company = String((JSON.parse(r.meta_json ?? "{}") as Record).recruitee_company ?? ""); } catch { /* skip */ } if (!company) continue; + seeded += 1; const url = `https://${encodeURIComponent(company)}.recruitee.com/api/offers/`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: RtResp = {}; try { parsed = JSON.parse(res.body) as RtResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "recruitee", res.body, "json"); @@ -70,7 +73,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `recruitee_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/smartrecruiters.ts b/apps/worker/src/prospects/sources/smartrecruiters.ts index f775cb28..c5dfa103 100644 --- a/apps/worker/src/prospects/sources/smartrecruiters.ts +++ b/apps/worker/src/prospects/sources/smartrecruiters.ts @@ -38,13 +38,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%smartrecruiters_company%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let slug = ""; try { slug = String((JSON.parse(r.meta_json ?? "{}") as Record).smartrecruiters_company ?? ""); } catch { /* skip */ } if (!slug) continue; + seeded += 1; const url = `https://api.smartrecruiters.com/v1/companies/${encodeURIComponent(slug)}/postings?limit=100`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: SrResp = {}; try { parsed = JSON.parse(res.body) as SrResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "smartrecruiters", res.body, "json"); @@ -69,7 +72,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `smartrecruiters_company` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/prospects/sources/workable.ts b/apps/worker/src/prospects/sources/workable.ts index 76172894..50545873 100644 --- a/apps/worker/src/prospects/sources/workable.ts +++ b/apps/worker/src/prospects/sources/workable.ts @@ -37,13 +37,16 @@ const mod: SourceModule = { : await ctx.env.DB.prepare( `SELECT id, domain, meta_json, name FROM accounts WHERE meta_json LIKE '%workable_account%' LIMIT 100`, ).all(); + let seeded = 0, boardsFetched = 0; for (const r of rows.results ?? []) { let slug = ""; try { slug = String((JSON.parse(r.meta_json ?? "{}") as Record).workable_account ?? ""); } catch { /* skip */ } if (!slug) continue; + seeded += 1; const url = `https://apply.workable.com/api/v3/accounts/${encodeURIComponent(slug)}/jobs`; const res = await compliantFetch(ctx.env, url, mod.slug, { accept: "application/json" }); if (!res || !res.ok) continue; + boardsFetched += 1; let parsed: WkResp = {}; try { parsed = JSON.parse(res.body) as WkResp; } catch { continue; } const r2_key = await archiveRaw(ctx.env, "workable", res.body, "json"); @@ -68,7 +71,16 @@ const mod: SourceModule = { }); } } - return { events, cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor }; + return { + events, + cursor: newest > since ? new Date(newest).toISOString() : ctx.cursor, + // Without this the run records `0 events, ok` whether it scanned a + // hundred boards and found nothing new or found nothing to scan at + // all. `workable_account` is operator-seeded on accounts.meta_json and + // nothing sets it automatically, so seeded_accounts: 0 is the normal + // state today — and the state an operator has no other way to see. + meta: { seeded_accounts: seeded, boards_fetched: boardsFetched }, + }; }, }; diff --git a/apps/worker/src/routes/leads.ts b/apps/worker/src/routes/leads.ts index f3561307..32efebe4 100644 --- a/apps/worker/src/routes/leads.ts +++ b/apps/worker/src/routes/leads.ts @@ -81,7 +81,19 @@ leads.get("/", async (c) => { `SELECT ${RICH_COLUMNS}, (SELECT json_group_array(r.role) FROM entity_legacy_map m JOIN entity_roles r ON r.entity_id = m.entity_id - WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id) AS roles_json + WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id) AS roles_json, + -- The unified id for this lead, when one exists. The Leads list + -- linked each name to /dashboard/people/?id=, but that + -- page feeds the value straight to /api/profilers/:entity_id/*, + -- which is keyed on u_entities. A legacy integer id never matches + -- a uuid, so every name on the page opened an empty profile. The + -- list already joins entity_legacy_map for the role badges, so + -- carrying the id costs one more correlated subquery and lets the + -- page link to a person page that resolves — or fall back to the + -- lead detail page for rows that have no entity yet. + (SELECT m.entity_id FROM entity_legacy_map m + WHERE m.legacy_table = 'leads' AND m.legacy_id = leads.id + LIMIT 1) AS entity_id FROM leads ${whereSql} ${orderSql} LIMIT ?`, ).bind(...binds, limit); const r = await stmt.all(); diff --git a/apps/worker/src/services/diligence/checks/founders.ts b/apps/worker/src/services/diligence/checks/founders.ts index ffa1a69d..b633a694 100644 --- a/apps/worker/src/services/diligence/checks/founders.ts +++ b/apps/worker/src/services/diligence/checks/founders.ts @@ -10,10 +10,29 @@ async function getFoundersOf(env: import("../../../types").Env, companyEntityId: // founders mirrored as facts (founder.company_founded) or via career role 'founder' const out = new Set(); try { + // The name fallback is not belt-and-braces: it is the only branch that + // currently matches. The single writer of this predicate — the founder + // profile workflow (crawler/profileWorkflows/founder.ts) — stores the + // company as free text in value_text, because at extraction time it has a + // company NAME off a bio page and no resolved entity. Matching only on + // value_entity_id therefore returned nothing, every time, and the whole + // founder-diligence section quietly fell back to the career_history path. + // The value_entity_id branch is kept first because it is the correct + // shape and will match once an extractor resolves the company. const r = await env.DB.prepare( - `SELECT entity_id FROM facts - WHERE predicate = 'founder.company_founded' AND value_entity_id = ? AND is_current = 1`, - ).bind(companyEntityId).all<{ entity_id: string }>(); + `SELECT f.entity_id FROM facts f + WHERE f.predicate = 'founder.company_founded' + AND f.is_current = 1 + AND ( + f.value_entity_id = ? + OR (f.value_entity_id IS NULL + AND f.value_text IS NOT NULL + AND TRIM(f.value_text) <> '' + AND LOWER(TRIM(f.value_text)) = ( + SELECT LOWER(TRIM(u.display_name)) FROM u_entities u WHERE u.id = ? + )) + )`, + ).bind(companyEntityId, companyEntityId).all<{ entity_id: string }>(); for (const row of r.results ?? []) out.add(row.entity_id); } catch { /* table may differ */ } try { diff --git a/apps/worker/src/services/edgeQuality/sweep.ts b/apps/worker/src/services/edgeQuality/sweep.ts index aa0436dd..b19e5a4c 100644 --- a/apps/worker/src/services/edgeQuality/sweep.ts +++ b/apps/worker/src/services/edgeQuality/sweep.ts @@ -15,6 +15,9 @@ import { collectAllSignals } from "./signals"; import { aggregateSignals } from "./aggregate"; import { computeInfluence, type ScoredEdge } from "./influence"; import { insertFact } from "../../entities/facts"; +import { firstOfJsonArray } from "../../entities/sector"; +import { logError } from "../../db/error_log"; +import { wrapUnknown } from "../../errors"; const EDGE_BATCH = 200; // edges scored per loop iteration const EDGE_TICK_CAP = 5000; // hard ceiling per nightly tick (Task #2 precedent) @@ -298,10 +301,23 @@ async function rebuildEntityInfluence(env: Env): Promise { } /** - * Best-effort sector lookup. Reads the most recent fact with predicate - * `entity.primary_sector` for each id; falls back to `firm.sector` for - * firm entities and `company.sector` for company entities. Empty for - * ids with no sector evidence. + * Best-effort sector lookup, driving the per-sector PageRank partition. + * + * This used to read only `entity.primary_sector`, `firm.sector` and + * `company.sector` from `facts`. No writer in the worker produces any of + * those three — the plural `firm.sectors` is what the profile workflows emit, + * `industry` is what the account dual-write emits, and both land as tags that + * the summary rebuild materialises into `entity_summary.sectors_csv`. So the + * map came back empty on every sweep, every node fell into the same unsectored + * bucket, and `sectors_ranked` was 0: per-sector PageRank and the per-sector + * power-node flags did nothing at all, silently, because an empty map is also + * what a genuinely unsectored graph produces. + * + * `entity_summary.sectors_csv` is now the primary source because it is the + * materialised one — already deduped, already slugged, one row per entity. + * The facts lookup stays as a fallback for entities whose summary has not been + * rebuilt yet, widened to the predicates that are actually written and reading + * `value_json` too, since the plural forms are stored as JSON arrays. */ async function loadPrimarySectors(env: Env, ids: string[]): Promise> { const out = new Map(); @@ -310,16 +326,38 @@ async function loadPrimarySectors(env: Env, ids: string[]): Promise '' + AND entity_id IN (${slice.map(() => "?").join(",")})`, + ).bind(...slice).all<{ entity_id: string; sectors_csv: string | null }>(); + for (const row of r.results ?? []) { + const first = (row.sectors_csv ?? "").split(",").map((x) => x.trim()).find(Boolean); + if (first && !out.has(row.entity_id)) out.set(row.entity_id, first.toLowerCase()); + } + } catch (e) { + await logError(env, { + err: wrapUnknown(e, "db_error", { chunk_start: i, chunk_size: slice.length }), + step: "edge_quality.primary_sector_summary", + }); + } + + const pending = slice.filter((id) => !out.has(id)); + if (!pending.length) continue; + try { + const r = await env.DB.prepare( + `SELECT entity_id, value_text, value_json FROM facts WHERE is_current = 1 - AND predicate IN ('entity.primary_sector','firm.sector','company.sector') - AND entity_id IN (${slice.map(() => "?").join(",")})`, - ).bind(...slice).all<{ entity_id: string; value_text: string | null }>(); + AND predicate IN ('entity.primary_sector','firm.sector','company.sector', + 'firm.sectors','company.sectors','sector','industry', + 'firm.industry') + AND entity_id IN (${pending.map(() => "?").join(",")})`, + ).bind(...pending).all<{ entity_id: string; value_text: string | null; value_json: string | null }>(); for (const row of r.results ?? []) { - if (row.value_text && !out.has(row.entity_id)) { - out.set(row.entity_id, row.value_text.toLowerCase()); - } + if (out.has(row.entity_id)) continue; + const v = row.value_text?.trim() || firstOfJsonArray(row.value_json); + if (v) out.set(row.entity_id, v.toLowerCase()); } } catch (e) { console.warn("primary sector lookup chunk failed", (e as Error).message); @@ -328,6 +366,7 @@ async function loadPrimarySectors(env: Env, ids: string[]): Promise '' + UNION ALL + -- Preference 2: the page the firm's people were actually scraped from. + -- scraper/pipeline.ts stamps firm_people.source_url with the team page + -- it parsed, which is exactly the URL this sweep wants to re-fetch. It + -- is the only populated source of a firm team URL in the schema. + -- CAST matches how portfolioFromFirmSite.ts and roleInference.ts join + -- firms into entity_legacy_map: firms.id is INTEGER, legacy_id is TEXT. + SELECT m.entity_id AS firm_entity_id, fp.source_url AS team_url, + 2 AS pref, fp.created_at AS ts_a, fp.created_at AS ts_b, CAST(fp.id AS TEXT) AS tie + FROM firm_people fp + JOIN entity_legacy_map m ON m.legacy_table = 'firms' + AND m.legacy_id = CAST(fp.firm_id AS TEXT) + WHERE fp.source_url IS NOT NULL + AND fp.source_url <> '' + ), + picked AS ( + SELECT c.firm_entity_id, c.team_url, + ROW_NUMBER() OVER ( + PARTITION BY c.firm_entity_id + ORDER BY c.pref ASC, c.ts_a DESC, c.ts_b DESC, c.tie ASC + ) AS rn + FROM candidates c + -- Widened from role = 'investor_firm', which only deals/investorResolver + -- assigns. Every firm that got its role from the dual-write carries + -- 'firm', so the old filter excluded the bulk of the table. This is + -- the same set the sibling detector (movements/spinout.ts) uses, and + -- the two reading the same graph differently was itself the bug. + JOIN entity_roles r ON r.entity_id = c.firm_entity_id + AND r.role IN ('investor_firm','firm','fund','vc','gp','investor') ), dated AS ( SELECT p.firm_entity_id, p.team_url, diff --git a/apps/worker/src/services/personaMatchTrigger.ts b/apps/worker/src/services/personaMatchTrigger.ts index 7b862f35..1b66a711 100644 --- a/apps/worker/src/services/personaMatchTrigger.ts +++ b/apps/worker/src/services/personaMatchTrigger.ts @@ -16,6 +16,9 @@ const RELEVANT_PREDICATES = new Set([ "person.location.country", "person.location.city", "location.country", "location.city", "employer", "person.employer", + // `employees` is the spelling that actually gets written; without it a + // fresh headcount fact never re-scored the entity it belongs to. + "employees", "org.headcount", "org.employees", "company.employees", "company.headcount", "org.sector", "sector", diff --git a/apps/worker/src/services/personaMatching.ts b/apps/worker/src/services/personaMatching.ts index a9408c05..5b5242cb 100644 --- a/apps/worker/src/services/personaMatching.ts +++ b/apps/worker/src/services/personaMatching.ts @@ -65,8 +65,19 @@ async function loadEmployerFacts(env: Env, employerId: string): Promise<{ const sum = await env.DB.prepare( `SELECT display_name, country_iso2, sectors_csv, stages_csv FROM entity_summary WHERE entity_id = ?`, ).bind(employerId).first<{ display_name: string | null; country_iso2: string | null; sectors_csv: string | null; stages_csv: string | null }>(); + // `employees` is first because it is the only one of these that anything + // writes: it is the predicate the registry declares + // (entities/profile-predicates.ts) and the one secEdgar/persist.ts and the + // account dual-write emit. The four `org.*` / `company.*` spellings below + // were the entire list, and no writer has ever produced one — so + // `employees` came back null for every entity and scoreCompanySize + // returned its "company size unknown" zero every time. With a weight of + // 0.10 that put a hard ceiling of 0.90 on every persona match, and made + // "company size unknown" a permanent line in the rationale the dashboard + // shows to explain why someone matched. The unused spellings are kept so a + // future writer picking one still resolves. const hc = await env.DB.prepare( - `SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`, + `SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('employees','org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`, ).bind(employerId).first<{ value_number: number | null }>(); if (!sum && !hc) return null; return { diff --git a/apps/worker/src/services/valuation/compPanel.ts b/apps/worker/src/services/valuation/compPanel.ts index 63e2722a..59cf3329 100644 --- a/apps/worker/src/services/valuation/compPanel.ts +++ b/apps/worker/src/services/valuation/compPanel.ts @@ -14,6 +14,7 @@ import type { Env } from "../../types"; import type { CompPanelCriteria } from "./types"; +import { entityHasSector, SECTOR_MATCHES_COMPANY_ENTITY_SQL } from "../../entities/sector"; export function parseCriteria(raw: string | null): CompPanelCriteria { if (!raw) return {}; @@ -78,14 +79,18 @@ export async function screenPanel( } // Sector / business_model: match against facts on the entity. if (criteria.sector) { - const f = await env.DB.prepare( - `SELECT 1 FROM facts WHERE entity_id = ? AND predicate IN ('company.sector','firm.sector','sector') - AND lower(value_text) = lower(?) AND is_current = 1 LIMIT 1`, - ).bind(r.company_entity_id, criteria.sector).first(); - if (!f) continue; + // A miss here `continue`s past the candidate, so reading the three + // singular predicates nothing writes did not return an unfiltered + // panel — it returned an EMPTY one, and an operator filtering by + // sector concluded there were no comparable companies. + if (!(await entityHasSector(env, r.company_entity_id, criteria.sector))) continue; reasons.push(`sector=${criteria.sector}`); } if (criteria.business_model) { + // Left as-is deliberately: `company.business_model` also has no writer, + // but unlike sector there is no alternative spelling anywhere in the + // schema to converge on. Widening it would be inventing a source. The + // criterion is operator-set, so it only bites when explicitly asked for. const f = await env.DB.prepare( `SELECT 1 FROM facts WHERE entity_id = ? AND predicate IN ('company.business_model','business_model') AND lower(value_text) = lower(?) AND is_current = 1 LIMIT 1`, @@ -111,12 +116,11 @@ export async function screenPanel( `SELECT DISTINCT vm.company_entity_id, e.display_name FROM valuation_marks vm JOIN u_entities e ON e.id = vm.company_entity_id - JOIN facts f ON f.entity_id = vm.company_entity_id - WHERE f.predicate IN ('company.sector','firm.sector','sector') - AND lower(f.value_text) = lower(?) AND f.is_current = 1 + WHERE ${SECTOR_MATCHES_COMPANY_ENTITY_SQL} AND vm.company_entity_id NOT IN (SELECT company_entity_id FROM comp_metrics WHERE ticker IS NOT NULL) LIMIT 500`, - ).bind(criteria.sector).all<{ company_entity_id: string; display_name: string }>(); + ).bind(criteria.sector, criteria.sector, criteria.sector) + .all<{ company_entity_id: string; display_name: string }>(); for (const r of (privRows.results ?? [])) { priv.push({ company_entity_id: r.company_entity_id, company_name: r.display_name, diff --git a/apps/worker/src/services/valuation/impliedValuation.ts b/apps/worker/src/services/valuation/impliedValuation.ts index 09d54cdb..844a6ac5 100644 --- a/apps/worker/src/services/valuation/impliedValuation.ts +++ b/apps/worker/src/services/valuation/impliedValuation.ts @@ -19,6 +19,7 @@ // selection can be extended to prefer panels whose stage matches the // target's most recent funding round. import type { Env } from "../../types"; +import { entityPrimarySector } from "../../entities/sector"; import type { ImpliedValuationRange } from "./types"; function percentile(sorted: number[], p: number): number { @@ -27,13 +28,11 @@ function percentile(sorted: number[], p: number): number { return sorted[idx]; } +// Delegates to entities/sector.ts. The three predicates this used to read +// have no writer anywhere in the worker, so it returned null for every +// company and the sector-matched panel fallback below never fired. async function getCompanySector(env: Env, entityId: string): Promise { - const r = await env.DB.prepare( - `SELECT value_text FROM facts - WHERE entity_id = ? AND predicate IN ('company.sector','firm.sector','sector') - AND is_current = 1 LIMIT 1`, - ).bind(entityId).first<{ value_text: string }>(); - return r?.value_text ?? null; + return entityPrimarySector(env, entityId); } async function pickPanelForCompany(env: Env, entityId: string): Promise<{ id: string; name: string; criteria_json: string } | null> { diff --git a/apps/worker/test-dist-q/ai/budget.js b/apps/worker/test-dist-q/ai/budget.js new file mode 100644 index 00000000..3cfcadb5 --- /dev/null +++ b/apps/worker/test-dist-q/ai/budget.js @@ -0,0 +1,46 @@ +// AI / Vectorize daily budget caps (Task #25 step 10). +// +// Reads AI_DAILY_NEURONS_CAP and VECTORIZE_DAILY_QUERIES_CAP from env vars. +// Polls the D1 ai_cost_daily roll-up (cached 60s in KV) for the running +// total; refuses calls past the cap so a runaway loop can't drain the +// account. /api/scrapers/health surfaces burn-down for the dashboard. +const KV_KEY = "ai-budget:today"; +const CACHE_TTL = 60; +export async function getBurn(env) { + const cached = await env.SCRAPE_CACHE?.get(KV_KEY); + if (cached) { + try { + return JSON.parse(cached); + } + catch { /* fall through */ } + } + const day = new Date().toISOString().slice(0, 10); + const totals = await env.DB.prepare(`SELECT + SUM(neurons) AS neurons, + SUM(cost_usd) AS cost, + SUM(CASE WHEN purpose LIKE 'vectorize_%' THEN calls ELSE 0 END) AS vec_calls + FROM ai_cost_daily WHERE day = ?`).bind(day).first().catch(() => null); + const snap = { + day, + neurons_used: Number(totals?.neurons ?? 0), + neurons_cap: Number(env.AI_DAILY_NEURONS_CAP ?? "0") || 0, + vectorize_used: Number(totals?.vec_calls ?? 0), + vectorize_cap: Number(env.VECTORIZE_DAILY_QUERIES_CAP ?? "0") || 0, + cost_usd: Number(totals?.cost ?? 0), + }; + try { + await env.SCRAPE_CACHE?.put(KV_KEY, JSON.stringify(snap), { expirationTtl: CACHE_TTL }); + } + catch { /* best-effort */ } + return snap; +} +export async function assertBudget(env, kind) { + const snap = await getBurn(env); + if (kind === "ai" && snap.neurons_cap > 0 && snap.neurons_used >= snap.neurons_cap) { + return { ok: false, reason: `neurons_cap_reached:${snap.neurons_used}/${snap.neurons_cap}` }; + } + if (kind === "vectorize" && snap.vectorize_cap > 0 && snap.vectorize_used >= snap.vectorize_cap) { + return { ok: false, reason: `vectorize_cap_reached:${snap.vectorize_used}/${snap.vectorize_cap}` }; + } + return { ok: true }; +} diff --git a/apps/worker/test-dist-q/ai/cache.js b/apps/worker/test-dist-q/ai/cache.js new file mode 100644 index 00000000..850329c6 --- /dev/null +++ b/apps/worker/test-dist-q/ai/cache.js @@ -0,0 +1,33 @@ +const PREFIX = "ai-cache"; +export async function sha256Hex(input) { + const buf = new TextEncoder().encode(input); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} +export async function aiCacheGet(env, key) { + if (!env.AI_CACHE) + return null; + try { + const obj = await env.AI_CACHE.get(`${PREFIX}/${key}`); + if (!obj) + return null; + const text = await obj.text(); + return JSON.parse(text); + } + catch { + return null; + } +} +export async function aiCachePut(env, key, value) { + if (!env.AI_CACHE) + return; + try { + await env.AI_CACHE.put(`${PREFIX}/${key}`, JSON.stringify(value), { + httpMetadata: { contentType: "application/json" }, + customMetadata: { stored_at: new Date().toISOString() }, + }); + } + catch { + /* swallow — cache is best-effort */ + } +} diff --git a/apps/worker/test-dist-q/ai/extract.js b/apps/worker/test-dist-q/ai/extract.js new file mode 100644 index 00000000..fdde5b67 --- /dev/null +++ b/apps/worker/test-dist-q/ai/extract.js @@ -0,0 +1,335 @@ +// AI-powered extraction (Task #25 step 2). +// +// Strategy: deterministic strategies in `firmcrawl/personExtract.ts` run +// first (cheap). Misses get routed through Workers AI with a strict JSON +// schema response, chunked at ~6KB. Every call is cached by +// sha256(model+prompt+chunk) in the AI_CACHE R2 bucket (30-day TTL via +// bucket lifecycle policy). Verification pass drops <0.6 confidence. +// +// Hooked into pipeline.ts firm_team_crawl path opportunistically: if `AI` +// binding is present and deterministic strategies returned <3 people on a +// non-trivial page, we run the AI pass and union by nameKey. +import { aiCacheGet, aiCachePut, sha256Hex } from "./cache"; +import { assertBudget } from "./budget"; +import { limitAi } from "../scraper/rateLimit"; +import { trackAi } from "../analytics/events"; +// Task #2: hard timeout for Workers AI calls. The binding does not accept +// AbortSignal, so we race against a timer and surface a uniform +// "ai_timeout" error. +// +// POLICY: one canonical 30s ceiling for any single AI call (constant +// `AI_TIMEOUT_MS` below). This is well below the 90s default job +// budget, so even three serial AI calls fit inside a single job's +// wall-clock ceiling. Short-form purposes (embeddings, arbitration) +// use `AI_TIMEOUT_SHORT_MS` (20s) since they're trivially smaller. +// +// NB: this is a *caller-side* timeout — it bounds how long the worker +// will wait on the Workers AI binding, but it does NOT cancel the +// underlying model execution. The binding doesn't expose an +// AbortSignal as of this revision, so model inference may continue +// (and bill) for a short tail after we move on. Acceptable today +// because the queue-level budget + sweeper will reclaim the job; if +// the binding gains cancellation, swap the race for a real abort. +const AI_TIMEOUT_MS = 30_000; +const AI_TIMEOUT_SHORT_MS = 20_000; +async function runAiWithTimeout(p, ms, label) { + let timer = null; + try { + return await Promise.race([ + p, + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error(`ai_timeout:${label}:${ms}ms`)), ms); + }), + ]); + } + finally { + if (timer) + clearTimeout(timer); + } +} +const PERSON_SCHEMA = { + type: "object", + properties: { + people: { + type: "array", + items: { + type: "object", + properties: { + name: { type: "string" }, + role: { type: "string" }, + email: { type: "string" }, + linkedin: { type: "string" }, + twitter: { type: "string" }, + bio: { type: "string" }, + confidence: { type: "number" }, + }, + required: ["name", "confidence"], + }, + }, + }, + required: ["people"], +}; +const CHUNK_BYTES = 6000; +const MIN_CONFIDENCE = 0.6; +function chunk(text, size) { + const out = []; + for (let i = 0; i < text.length; i += size) + out.push(text.slice(i, i + size)); + return out; +} +function stripHtml(html) { + return html + .replace(/]*>[\s\S]*?<\/script>/gi, " ") + .replace(/]*>[\s\S]*?<\/style>/gi, " ") + .replace(/<[^>]+>/g, " ") + .replace(/\s+/g, " ") + .trim(); +} +export async function aiExtractPeople(env, html, jobId) { + if (!env.AI) + return []; + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return []; + if (!(await limitAi(env))) + return []; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const text = stripHtml(html); + const chunks = chunk(text, CHUNK_BYTES).slice(0, 4); // hard ceiling per page + const all = []; + for (const c of chunks) { + const cacheKey = await sha256Hex(`${model}:people:${c}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "extraction", model, cacheHit: true, jobId }); + all.push(...cached); + continue; + } + const t0 = Date.now(); + let people = []; + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "Extract investors/partners as JSON. Skip non-people. Return strict JSON." }, + { role: "user", content: `Extract people from this team-page text. ${c}` }, + ], + response_format: { type: "json_schema", json_schema: PERSON_SCHEMA }, + }), AI_TIMEOUT_MS, "extract_people")); + const parsed = parsePeopleResponse(res); + people = parsed.filter((p) => (p.confidence ?? 0) >= MIN_CONFIDENCE); + } + catch (e) { + console.warn("aiExtractPeople failed", e.message); + } + trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(c.length), jobId }); + await aiCachePut(env, cacheKey, people); + all.push(...people); + } + return dedupePeopleByName(all); +} +function parsePeopleResponse(res) { + const r = res; + if (Array.isArray(r?.people)) + return r.people.filter((p) => p && typeof p.name === "string"); + if (typeof r?.response === "string") { + try { + const j = JSON.parse(r.response); + if (Array.isArray(j?.people)) + return j.people.filter((p) => p && typeof p.name === "string"); + } + catch { /* fall through */ } + } + return []; +} +function dedupePeopleByName(arr) { + const map = new Map(); + for (const p of arr) { + const key = p.name.trim().toLowerCase(); + if (!key) + continue; + const cur = map.get(key); + if (!cur || (p.confidence ?? 0) > (cur.confidence ?? 0)) + map.set(key, p); + } + return [...map.values()]; +} +// Rough neurons estimate: tokens ≈ chars/4, llama-3.1-8b ~ 0.011 neurons/token. +function estimateNeurons(chars) { + const tokens = Math.ceil(chars / 4); + return Math.round(tokens * 0.011 * 1000) / 1000; +} +const TABLE_SCHEMA = { + type: "object", + properties: { + tables: { + type: "array", + items: { + type: "object", + properties: { + headers: { type: "array", items: { type: "string" } }, + rows: { type: "array", items: { type: "array", items: { type: "string" } } }, + }, + required: ["headers", "rows"], + }, + }, + }, + required: ["tables"], +}; +const TABLE_PAGE_CHAR_CAP = 8000; +const MAX_AI_PAGES = 12; +export async function aiExtractTablesFromPdfPages(env, pageTexts) { + if (!env.AI || !pageTexts.length) + return []; + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return []; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const out = []; + let lastHeaderKey = null; + const pages = pageTexts.slice(0, MAX_AI_PAGES); + for (let p = 0; p < pages.length; p++) { + const raw = pages[p].trim(); + if (raw.length < 40) + continue; + const text = raw.length > TABLE_PAGE_CHAR_CAP ? raw.slice(0, TABLE_PAGE_CHAR_CAP) : raw; + const cacheKey = await sha256Hex(`${model}:pdf-tables:${text}`); + let pageTables = await aiCacheGet(env, cacheKey); + if (pageTables) { + trackAi(env, { purpose: "extraction", model, cacheHit: true }); + } + else { + if (!(await limitAi(env))) + continue; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "You extract tabular data from a single PDF page. Return strict JSON {tables: [{headers, rows}]}. Skip page numbers, app chrome (File/Edit/View toolbars, sheet tab strips, Share buttons), and prose paragraphs. If the page has no table, return {tables: []}. Each row must have the same length as headers; pad with empty strings if needed." }, + { role: "user", content: `PDF page text:\n${text}` }, + ], + response_format: { type: "json_schema", json_schema: TABLE_SCHEMA }, + }), AI_TIMEOUT_MS, "extract_tables")); + pageTables = parseTablesResponse(res); + } + catch (e) { + console.warn("aiExtractTablesFromPdfPages failed", e.message); + pageTables = []; + } + trackAi(env, { purpose: "extraction", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); + await aiCachePut(env, cacheKey, pageTables); + } + for (const t of pageTables) { + if (!Array.isArray(t.headers) || t.headers.length < 2) + continue; + if (!Array.isArray(t.rows) || t.rows.length < 1) + continue; + const headers = t.headers.map((h) => String(h || "").trim()); + const headerKey = headers.join("|").toLowerCase(); + const rows = t.rows.map((r) => { + const obj = {}; + for (let c = 0; c < headers.length; c++) + obj[headers[c] || `col_${c}`] = String(r?.[c] ?? "").trim(); + return obj; + }).filter((r) => Object.values(r).some((v) => v.length > 0)); + if (!rows.length) + continue; + if (lastHeaderKey === headerKey && out.length) { + out[out.length - 1].rows.push(...rows); + } + else { + out.push({ headers, rows, pageNumber: p + 1, confidence: 0.5 }); + lastHeaderKey = headerKey; + } + } + } + return out; +} +function parseTablesResponse(res) { + const r = res; + if (Array.isArray(r?.tables)) + return r.tables; + if (typeof r?.response === "string") { + try { + const j = JSON.parse(r.response); + if (Array.isArray(j?.tables)) + return j.tables; + } + catch { /* fall through */ } + } + return []; +} +export async function aiEmbed(env, text) { + if (!env.AI) + return null; + const model = env.AI_EMBED_MODEL ?? "@cf/baai/bge-base-en-v1.5"; + const cacheKey = await sha256Hex(`${model}:embed:${text}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "embedding", model, cacheHit: true }); + return cached; + } + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return null; + if (!(await limitAi(env))) + return null; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { text: [text] }), AI_TIMEOUT_SHORT_MS, "embed")); + const vec = Array.isArray(res?.data?.[0]) ? res.data[0] : null; + if (!vec) + return null; + trackAi(env, { purpose: "embedding", model, ms: Date.now() - t0, neurons: estimateNeurons(text.length) }); + await aiCachePut(env, cacheKey, vec); + return vec; + } + catch (e) { + console.warn("aiEmbed failed", e.message); + return null; + } +} +export async function aiArbitrate(env, candidateA, candidateB) { + if (!env.AI) + return { match: "maybe", confidence: 0 }; + const model = env.AI_EXTRACT_MODEL ?? "@cf/meta/llama-3.1-8b-instruct-fast"; + const cacheKey = await sha256Hex(`${model}:arb:${candidateA}|${candidateB}`); + const cached = await aiCacheGet(env, cacheKey); + if (cached) { + trackAi(env, { purpose: "arbitration", model, cacheHit: true }); + return cached; + } + const ok = await assertBudget(env, "ai"); + if (!ok.ok) + return { match: "maybe", confidence: 0 }; + if (!(await limitAi(env))) + return { match: "maybe", confidence: 0 }; + const t0 = Date.now(); + try { + const res = (await runAiWithTimeout(env.AI.run(model, { + messages: [ + { role: "system", content: "Decide if two profiles describe the same person. Reply JSON: {match: yes|no|maybe, confidence: 0..1}." }, + { role: "user", content: `A: ${candidateA}\nB: ${candidateB}` }, + ], + response_format: { type: "json_object" }, + }), AI_TIMEOUT_SHORT_MS, "arbitrate")); + const out = parseArbResponse(res); + trackAi(env, { purpose: "arbitration", model, ms: Date.now() - t0, neurons: estimateNeurons(candidateA.length + candidateB.length) }); + await aiCachePut(env, cacheKey, out); + return out; + } + catch (e) { + console.warn("aiArbitrate failed", e.message); + return { match: "maybe", confidence: 0 }; + } +} +function parseArbResponse(res) { + if (typeof res?.response === "string") { + try { + const j = JSON.parse(res.response); + const match = j.match === "yes" || j.match === "no" ? j.match : "maybe"; + return { match, confidence: Math.max(0, Math.min(1, Number(j.confidence ?? 0))) }; + } + catch { /* fall through */ } + } + return { match: "maybe", confidence: 0 }; +} diff --git a/apps/worker/test-dist-q/analytics/events.js b/apps/worker/test-dist-q/analytics/events.js new file mode 100644 index 00000000..c9fb991c --- /dev/null +++ b/apps/worker/test-dist-q/analytics/events.js @@ -0,0 +1,79 @@ +// Analytics Engine event helpers (Task #25 step 7). +// +// Cloudflare Analytics Engine accepts up to 20 blobs (strings), 20 doubles, +// and 1 list of indexes per data point. We standardize the schema across all +// AI/fetch events so the GraphQL Analytics API queries stay simple. +// +// Schema: +// indexes: [purpose] e.g. "extraction" | "embedding" | "arbitration" | … +// blobs: [purpose, model, host, cache_hit, job_id] +// doubles: [neurons, ms, bytes, cost_usd] +export function trackAi(env, args) { + const purpose = args.purpose; + const cache = args.cacheHit ? "1" : "0"; + try { + env.ANALYTICS?.writeDataPoint({ + indexes: [purpose], + blobs: [purpose, args.model, "", cache, args.jobId ?? ""], + doubles: [args.neurons ?? 0, args.ms ?? 0, 0, args.costUsd ?? 0], + }); + } + catch { + /* analytics is best-effort */ + } + // Also persist a daily roll-up to D1 so /api/analytics/ae/ai-cost works + // without the GraphQL Analytics API (which requires an account-level token). + if (!args.cacheHit) { + void rollupAiCost(env, purpose, args.model, args.neurons ?? 0, args.costUsd ?? 0); + } +} +// Vectorize ops are billed per query/upsert independent of AI neurons. We +// count them in the same ai_cost_daily roll-up under purpose='vectorize_' +// so /api/scrapers/health can surface the daily burn against +// VECTORIZE_DAILY_QUERIES_CAP. +export function trackVectorize(env, args) { + try { + env.ANALYTICS?.writeDataPoint({ + indexes: [`vectorize_${args.op}`], + blobs: [`vectorize_${args.op}`, args.index, "", "0", ""], + doubles: [0, 0, 0, 0], + }); + } + catch { /* best-effort */ } + void rollupVectorizeOp(env, args.op, args.index); +} +async function rollupVectorizeOp(env, op, index) { + try { + const day = new Date().toISOString().slice(0, 10); + await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) + VALUES (?, ?, ?, 0, 0, 1) + ON CONFLICT(day, purpose, model) DO UPDATE SET calls = calls + 1`).bind(day, `vectorize_${op}`, index).run(); + } + catch (e) { + console.warn("vectorize rollup failed", e.message); + } +} +export function trackFetch(env, args) { + try { + env.ANALYTICS?.writeDataPoint({ + indexes: ["fetch"], + blobs: ["fetch", String(args.tier), args.host, args.blockReason ?? "", String(args.status)], + doubles: [0, args.ms, args.bytes, 0], + }); + } + catch { /* best-effort */ } +} +async function rollupAiCost(env, purpose, model, neurons, costUsd) { + try { + const day = new Date().toISOString().slice(0, 10); + await env.DB.prepare(`INSERT INTO ai_cost_daily (day, purpose, model, neurons, cost_usd, calls) + VALUES (?, ?, ?, ?, ?, 1) + ON CONFLICT(day, purpose, model) DO UPDATE SET + neurons = neurons + excluded.neurons, + cost_usd = cost_usd + excluded.cost_usd, + calls = calls + 1`).bind(day, purpose, model, neurons, costUsd).run(); + } + catch (e) { + console.warn("ai_cost_daily rollup failed", e.message); + } +} diff --git a/apps/worker/test-dist-q/db/error_log.js b/apps/worker/test-dist-q/db/error_log.js new file mode 100644 index 00000000..d4ae5fec --- /dev/null +++ b/apps/worker/test-dist-q/db/error_log.js @@ -0,0 +1,125 @@ +// Task #27: structured error logging into D1 + Analytics Engine mirror. +// +// Best-effort writes — we never let a logging failure mask the original +// error. Truncation rules keep us under D1's 1MB row cap. Every call also +// emits one Analytics Engine data point so spike detection / alerting can +// be done without scanning D1. +import { AppError, wrapUnknown } from "../errors"; +const MAX_MSG = 2000; +const MAX_STACK = 8000; +const MAX_CONTEXT_JSON = 16000; +function clip(s, max) { + if (!s) + return null; + return s.length > max ? s.slice(0, max) + "…[clipped]" : s; +} +function hostFromUrl(u) { + if (!u) + return null; + try { + return new URL(u).hostname.toLowerCase(); + } + catch { + return null; + } +} +export async function logError(env, input) { + const e = input.err instanceof AppError ? input.err : wrapUnknown(input.err, "internal_error"); + const host = input.host ?? hostFromUrl(input.url); + // Mirror to Analytics Engine first (cheap, doesn't depend on D1). + try { + if (env.ANALYTICS) { + env.ANALYTICS.writeDataPoint({ + indexes: [e.code], + blobs: [ + e.kind, + e.code, + input.step ?? "", + input.job_id ?? "", + input.request_id ?? "", + host ?? "", + input.workflow_run_id ?? "", + input.user_email ?? "", + ], + doubles: [e.status, e.retryable ? 1 : 0, input.retry_count ?? 0], + }); + } + } + catch { /* never throw from logger */ } + if (!env.DB) + return null; + let contextJson = null; + try { + contextJson = clip(JSON.stringify(e.context ?? {}), MAX_CONTEXT_JSON); + } + catch { + contextJson = null; + } + try { + const r = await env.DB.prepare(`INSERT INTO error_log + (request_id, job_id, step, code, kind, status, retryable, message, context_json, + cause_name, cause_message, cause_stack, url, method, + workflow_run_id, host, user_email, retry_count) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) + .bind(input.request_id ?? null, input.job_id ?? null, input.step ?? null, e.code, e.kind, e.status, e.retryable ? 1 : 0, clip(e.message, MAX_MSG), contextJson, e.cause?.name ?? null, clip(e.cause?.message, MAX_MSG), clip(e.cause?.stack, MAX_STACK), input.url ?? null, input.method ?? null, input.workflow_run_id ?? null, host, input.user_email ?? null, input.retry_count ?? 0) + .run(); + const id = r.meta?.last_row_id; + return typeof id === "number" ? id : null; + } + catch (logErr) { + // Never throw from the logger. + // Telemetry-of-telemetry: never throw, never log to console (CI gate). + void logErr; + return null; + } +} +export async function logStep(env, input) { + if (!env.DB) + return; + let metaJson = null; + try { + metaJson = input.meta ? JSON.stringify(input.meta) : null; + } + catch { + metaJson = null; + } + const finishedAt = input.status === "started" ? null : new Date().toISOString(); + try { + await env.DB.prepare(`INSERT INTO workflow_step_log + (job_id, step, step_name, status, finished_at, duration_ms, count_in, count_out, error_id, meta_json, workflow_run_id, attempt, error_code) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`) + .bind(input.job_id, input.step, input.step, input.status, finishedAt, input.duration_ms ?? null, input.count_in ?? null, input.count_out ?? null, input.error_id ?? null, metaJson, input.workflow_run_id ?? null, input.attempt ?? 1, input.error_code ?? null) + .run(); + } + catch (e) { + void e; + } +} +/** Convenience: time a function and log start+finish to workflow_step_log. */ +export async function timedStep(env, job_id, step, fn, opts = {}) { + const t0 = Date.now(); + const { workflow_run_id, attempt } = opts; + await logStep(env, { job_id, step, status: "started", count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt }); + try { + const out = await fn(); + const count_out = Array.isArray(out) ? out.length : undefined; + await logStep(env, { job_id, step, status: "ok", duration_ms: Date.now() - t0, count_in: opts.count_in, count_out, meta: opts.meta, workflow_run_id, attempt }); + return out; + } + catch (e) { + const { classify, isBenignSkip } = await import("../errors.js"); + // Task #72: robots.txt / ToS blocks are benign policy skips, not errors. + // Don't write an error_log row (it would surface as a red 422 in the + // operator console) — record the step as `skipped` instead and rethrow so + // the caller routes the job to the `skipped` terminal status. + const skip = isBenignSkip(e); + if (skip) { + await logStep(env, { job_id, step, status: "skipped", duration_ms: Date.now() - t0, count_in: opts.count_in, meta: opts.meta, workflow_run_id, attempt, error_code: skip.skip_code }); + throw e; + } + const error_id = await logError(env, { err: e, job_id, step }); + const cls = classify(e); + await logStep(env, { job_id, step, status: "error", duration_ms: Date.now() - t0, count_in: opts.count_in, error_id, meta: opts.meta, workflow_run_id, attempt, error_code: cls?.code }); + throw e; + } +} diff --git a/apps/worker/test-dist-q/entities/channels.js b/apps/worker/test-dist-q/entities/channels.js new file mode 100644 index 00000000..41a13c4c --- /dev/null +++ b/apps/worker/test-dist-q/entities/channels.js @@ -0,0 +1,61 @@ +// Channel upsert keyed by (entity_id, kind, canonical). Canonical form +// is computed by ./normalize so trivially-different inputs collapse. +import { canonicalEmail, canonicalPhone, canonicalLinkedin, canonicalTwitter, canonicalGithub, canonicalUrl, } from "./normalize"; +export function canonicalizeFor(kind, raw) { + switch (kind) { + case "email": return canonicalEmail(raw); + case "phone": return canonicalPhone(raw); + case "linkedin": return canonicalLinkedin(raw); + case "twitter": return canonicalTwitter(raw); + case "github": return canonicalGithub(raw); + case "website": + case "other": + return canonicalUrl(raw); + } +} +export async function upsertChannel(env, input) { + if (!input.entity_id || !input.canonical) + return null; + // Re-canonicalize defensively in case caller passed a raw value. + const canonical = canonicalizeFor(input.kind, input.canonical) ?? input.canonical; + const now = new Date().toISOString(); + const existing = await env.DB.prepare(`SELECT id FROM channels WHERE entity_id = ? AND kind = ? AND canonical = ?`).bind(input.entity_id, input.kind, canonical).first(); + if (existing) { + const sets = ["last_seen_at = ?"]; + const binds = [now]; + if (input.display) { + sets.push("display = COALESCE(display, ?)"); + binds.push(input.display); + } + if (input.is_primary) { + sets.push("is_primary = 1"); + } + if (input.is_verified) { + sets.push("is_verified = 1"); + } + if (input.is_dnc) { + sets.push("is_dnc = 1"); + } + if (input.source) { + sets.push("source = COALESCE(source, ?)"); + binds.push(input.source); + } + binds.push(existing.id); + await env.DB.prepare(`UPDATE channels SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); + return existing.id; + } + const id = crypto.randomUUID(); + await env.DB.prepare(`INSERT INTO channels (id, entity_id, kind, canonical, display, is_primary, is_verified, is_dnc, source, confidence, first_seen_at, last_seen_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`).bind(id, input.entity_id, input.kind, canonical, input.display ?? null, input.is_primary ? 1 : 0, input.is_verified ? 1 : 0, input.is_dnc ? 1 : 0, input.source ?? null, input.confidence ?? 1, now, now).run(); + return id; +} +export async function findEntityByChannel(env, kind, raw) { + const canonical = canonicalizeFor(kind, raw); + if (!canonical) + return null; + const r = await env.DB.prepare(`SELECT c.entity_id FROM channels c + JOIN u_entities e ON e.id = c.entity_id + WHERE c.kind = ? AND c.canonical = ? AND e.status NOT IN ('merged','soft_deleted') + ORDER BY c.is_primary DESC, c.is_verified DESC, c.last_seen_at DESC LIMIT 1`).bind(kind, canonical).first(); + return r?.entity_id ?? null; +} diff --git a/apps/worker/test-dist-q/entities/dualwrite.js b/apps/worker/test-dist-q/entities/dualwrite.js new file mode 100644 index 00000000..505a51bc --- /dev/null +++ b/apps/worker/test-dist-q/entities/dualwrite.js @@ -0,0 +1,375 @@ +// Dual-write hooks. Every legacy writer calls one of these to mirror the +// row into the unified entity model. All hooks are best-effort: they +// log + swallow errors so a transient unified-model failure never blocks +// the legacy path. +import { createEntity, addRole, getLegacyEntityId, setLegacyEntityId } from "./roles"; +import { insertFactsBatch } from "./facts"; +import { upsertChannel, findEntityByChannel } from "./channels"; +import { addTag, addTagsFromJsonArray } from "./tags"; +import { canonicalEmail, canonicalLinkedin, canonicalDomain } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +async function resolveOrCreate(env, table, legacyId, kind, init, channelLookups) { + const existing = await getLegacyEntityId(env, table, legacyId); + if (existing) + return existing; + // Cross-link by strongest available identifier before creating new: + // (a) deterministic domain match against u_entities.primary_domain + // (orgs only — collapses firm/account/company duplicates sharing a + // domain); (b) primary_email_key/primary_linkedin_key direct hits; + // (c) channel-table reverse lookups for any other handle. + if (kind === "org" && init.primary_domain) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_domain = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_domain).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + if (kind === "person" && init.primary_email_key) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_email_key = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_email_key).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + if (init.primary_linkedin_key) { + const r = await env.DB.prepare(`SELECT id FROM u_entities + WHERE primary_linkedin_key = ? AND status NOT IN ('merged','soft_deleted') + LIMIT 1`).bind(init.primary_linkedin_key).first(); + if (r?.id) { + await setLegacyEntityId(env, table, legacyId, r.id); + return r.id; + } + } + for (const ch of channelLookups) { + if (!ch.raw) + continue; + const hit = await findEntityByChannel(env, ch.kind, ch.raw); + if (hit) { + await setLegacyEntityId(env, table, legacyId, hit); + return hit; + } + } + try { + const created = await createEntity(env, { kind, ...init }); + if (!created) + return null; // Task #9: rejected by garbage detector + await setLegacyEntityId(env, table, legacyId, created.id); + return created.id; + } + catch (e) { + console.warn("dualwrite createEntity failed", table, legacyId, e.message); + return null; + } +} +function chan(env, entityId, kind, raw, source, primary = false) { + if (!raw) + return Promise.resolve(null); + return upsertChannel(env, { entity_id: entityId, kind, canonical: String(raw), source, is_primary: primary }); +} +export async function syncFirmToEntity(env, f, source = "firms_upsert", sourceKind = "scrape") { + try { + const domain = canonicalDomain(f.domain ?? f.website); + const linkedin = canonicalLinkedin(f.linkedin_url); + const entityId = await resolveOrCreate(env, "firms", f.id, "org", { + display_name: f.name, + primary_domain: domain, + primary_url: f.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: f.linkedin_url }, + { kind: "website", raw: f.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "firm", { is_primary: true, source }); + if (f.kind && /accelerator/i.test(f.kind)) + await addRole(env, entityId, "accelerator", { source }); + // Task #2: cascading role inference on the unified write path so + // investor_firm / customer / prospect get assigned without each + // call-site having to know. + try { + const { inferAndAssignRoles } = await import("../services/roleInference.js"); + await inferAndAssignRoles(env, entityId, { + kind: "org", + sourceKind, + sourceUrl: f.website ?? null, + sourceDomain: domain ?? null, + org: f.name ?? null, + category: f.kind ?? null, + importLabel: source, + }); + } + catch (e) { + console.warn("inferAndAssignRoles(firm) failed", entityId, e.message); + } + const patches = [ + { predicate: "name", value_text: f.name }, + { predicate: "legal_name", value_text: f.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: f.website ?? null }, + { predicate: "country_iso2", value_text: f.hq_country_iso2 ?? null }, + { predicate: "region", value_text: f.hq_region ?? null }, + { predicate: "city", value_text: f.hq_city ?? null }, + { predicate: "thesis", value_text: f.thesis ?? null }, + { predicate: "kind", value_text: f.kind ?? null }, + { predicate: "check_size_min_usd", value_number: numOrNull(f.check_size_min_usd) }, + { predicate: "check_size_max_usd", value_number: numOrNull(f.check_size_max_usd) }, + { predicate: "check_size_typical_usd", value_number: numOrNull(f.check_size_typical_usd) }, + ]; + await insertFactsBatch(env, entityId, patches, source, sourceKind); + await Promise.all([ + chan(env, entityId, "website", f.website, source, true), + chan(env, entityId, "linkedin", f.linkedin_url, source, true), + chan(env, entityId, "other", f.crunchbase_url, source), + chan(env, entityId, "twitter", f.twitter_handle, source), + chan(env, entityId, "email", f.contact_email, source), + ]); + await Promise.all([ + addTagsFromJsonArray(env, entityId, "sector", f.sectors_json, source), + addTagsFromJsonArray(env, entityId, "stage", f.stages_json, source), + addTagsFromJsonArray(env, entityId, "geo", f.geo_focus_json, source), + ]); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncFirmToEntity failed", f.id, e.message); + return null; + } +} +export async function syncLeadToEntity(env, l, source = "leads_repo", sourceKind = "scrape") { + try { + const email = canonicalEmail(l.email); + const linkedin = canonicalLinkedin(l.linkedin_url); + const entityId = await resolveOrCreate(env, "leads", l.id, "person", { + display_name: l.name ?? null, + primary_email_key: email, + primary_linkedin_key: linkedin, + primary_url: l.personal_url ?? null, + }, [ + { kind: "email", raw: l.email }, + { kind: "linkedin", raw: l.linkedin_url }, + ]); + if (!entityId) + return null; + // Role inference: investor_kind set → investor; otherwise generic person. + if (l.investor_kind) + await addRole(env, entityId, "investor", { is_primary: true, source }); + if (l.category && /founder|ceo|cto/i.test(l.category)) + await addRole(env, entityId, "founder", { source }); + // Task #2: cascading role inference on the unified write path. Runs + // for every person sync so the Investors page (which reads from + // entity_roles) picks up freshly-ingested people. Looks up the + // person's company entity for partner-title inheritance. + try { + const { inferAndAssignRoles } = await import("../services/roleInference.js"); + await inferAndAssignRoles(env, entityId, { + kind: "person", + sourceKind, + sourceUrl: l.source_url ?? l.personal_url ?? null, + sourceDomain: l.source_domain ?? null, + title: l.title ?? null, + org: l.org ?? null, + category: l.category ?? null, + importLabel: source, + }); + } + catch (e) { + console.warn("inferAndAssignRoles(lead) failed", entityId, e.message); + } + const patches = [ + { predicate: "name", value_text: l.name ?? null }, + { predicate: "title", value_text: l.title ?? null }, + { predicate: "primary_employer", value_text: l.org ?? null }, + { predicate: "category", value_text: l.category ?? null }, + { predicate: "country_iso2", value_text: l.country_iso2 ?? null }, + { predicate: "region", value_text: l.region ?? null }, + { predicate: "city", value_text: l.city ?? null }, + { predicate: "bio", value_text: l.bio ?? null }, + { predicate: "thesis", value_text: l.thesis ?? null }, + { predicate: "investor_kind", value_text: l.investor_kind ?? null }, + { predicate: "check_size_min_usd", value_number: numOrNull(l.check_size_min_usd) }, + { predicate: "check_size_max_usd", value_number: numOrNull(l.check_size_max_usd) }, + { predicate: "check_size_typical_usd", value_number: numOrNull(l.check_size_typical_usd) }, + ]; + await insertFactsBatch(env, entityId, patches, source, sourceKind); + await Promise.all([ + chan(env, entityId, "email", l.email, source, true), + chan(env, entityId, "phone", l.phone, source), + chan(env, entityId, "linkedin", l.linkedin_url, source, true), + chan(env, entityId, "twitter", l.twitter_url, source), + chan(env, entityId, "github", l.github_url, source), + chan(env, entityId, "website", l.personal_url, source), + ]); + await Promise.all([ + addTagsFromJsonArray(env, entityId, "sector", l.sector_focus_json, source), + addTagsFromJsonArray(env, entityId, "stage", l.stage_focus_json, source), + addTagsFromJsonArray(env, entityId, "geo", l.geo_focus_json, source), + addTagsFromJsonArray(env, entityId, "tag", l.tags_json, source), + ]); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncLeadToEntity failed", l.id, e.message); + return null; + } +} +export async function syncCompanyToEntity(env, c, source = "companies") { + try { + const domain = canonicalDomain(c.domain ?? c.website); + const linkedin = canonicalLinkedin(c.linkedin_url); + const entityId = await resolveOrCreate(env, "companies", c.id, "org", { + display_name: c.name, + primary_domain: domain, + primary_url: c.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: c.linkedin_url }, + { kind: "website", raw: c.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "company", { is_primary: true, source }); + const patches = [ + { predicate: "name", value_text: c.name }, + { predicate: "legal_name", value_text: c.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: c.website ?? null }, + { predicate: "country_iso2", value_text: c.hq_country_iso2 ?? null }, + { predicate: "region", value_text: c.hq_region ?? null }, + { predicate: "city", value_text: c.hq_city ?? null }, + { predicate: "stage", value_text: c.stage ?? null }, + { predicate: "unicorn_count", value_number: c.unicorn ? 1 : 0 }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "website", c.website, source, true), + chan(env, entityId, "linkedin", c.linkedin_url, source), + chan(env, entityId, "twitter", c.twitter_handle, source), + chan(env, entityId, "github", c.github_org, source), + chan(env, entityId, "other", c.crunchbase_url, source), + ]); + await addTagsFromJsonArray(env, entityId, "sector", c.industries_json, source); + if (c.stage) + await addTag(env, { entity_id: entityId, taxonomy: "stage", slug: c.stage, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncCompanyToEntity failed", c.id, e.message); + return null; + } +} +export async function syncAccountToEntity(env, a, source = "accounts") { + try { + const domain = canonicalDomain(a.domain ?? a.website); + const linkedin = canonicalLinkedin(a.linkedin_url); + const entityId = await resolveOrCreate(env, "accounts", a.id, "org", { + display_name: a.name, + primary_domain: domain, + primary_url: a.website ?? null, + primary_linkedin_key: linkedin, + }, [ + { kind: "linkedin", raw: a.linkedin_url }, + { kind: "website", raw: a.website ?? null }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "account", { is_primary: true, source }); + const patches = [ + { predicate: "name", value_text: a.name }, + { predicate: "legal_name", value_text: a.legal_name ?? null }, + { predicate: "domain", value_text: domain }, + { predicate: "website", value_text: a.website ?? null }, + { predicate: "country_iso2", value_text: a.hq_country_iso2 ?? null }, + { predicate: "region", value_text: a.hq_region ?? null }, + { predicate: "city", value_text: a.hq_city ?? null }, + { predicate: "industry", value_text: a.industry ?? null }, + { predicate: "funding_stage", value_text: a.funding_stage ?? null }, + { predicate: "fit_max_score", value_number: numOrNull(a.fit_score) }, + { predicate: "intent_score", value_number: numOrNull(a.intent_score) }, + // `accounts.employees` was the one sizing column dualwrite dropped, so + // no account entity carried a headcount fact and persona matching + // scored every one of them "company size unknown". `employees` is the + // predicate the registry declares (entities/profile-predicates.ts) and + // the one secEdgar/persist.ts already writes, so this converges on the + // name that exists rather than adding a fourth spelling. + { predicate: "employees", value_number: numOrNull(a.employees) }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "website", a.website, source, true), + chan(env, entityId, "linkedin", a.linkedin_url, source), + chan(env, entityId, "twitter", a.twitter_handle, source), + chan(env, entityId, "github", a.github_org, source), + chan(env, entityId, "other", a.crunchbase_url, source), + ]); + await addTagsFromJsonArray(env, entityId, "sector", a.industries_json, source); + if (a.industry) + await addTag(env, { entity_id: entityId, taxonomy: "sector", slug: a.industry, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncAccountToEntity failed", a.id, e.message); + return null; + } +} +export async function syncBuyerToEntity(env, b, source = "buyers") { + try { + const email = canonicalEmail(b.email); + const linkedin = canonicalLinkedin(b.linkedin_url); + const entityId = await resolveOrCreate(env, "buyers", b.id, "person", { + display_name: b.name ?? null, + primary_email_key: email, + primary_linkedin_key: linkedin, + }, [ + { kind: "email", raw: b.email }, + { kind: "linkedin", raw: b.linkedin_url }, + ]); + if (!entityId) + return null; + await addRole(env, entityId, "buyer", { is_primary: true, source }); + if (b.is_decision_maker) + await addRole(env, entityId, "executive", { source }); + // Link buyer → account via 'works_at' edge. + const accountEntityId = await getLegacyEntityId(env, "accounts", b.account_id); + if (accountEntityId) { + await env.DB.prepare(`INSERT INTO rel_edges (id, src_entity_id, dst_entity_id, kind, source) + VALUES (?, ?, ?, 'works_at', ?) + ON CONFLICT(src_entity_id, dst_entity_id, kind, IFNULL(valid_from,'')) DO NOTHING`).bind(crypto.randomUUID(), entityId, accountEntityId, source).run().catch(() => undefined); + // Also write as a fact for summary.primary_employer_entity_id pickup. + await insertFactsBatch(env, entityId, [{ predicate: "employer", value_entity_id: accountEntityId }], source, "inferred"); + } + const patches = [ + { predicate: "name", value_text: b.name ?? null }, + { predicate: "title", value_text: b.title ?? null }, + { predicate: "seniority", value_text: b.seniority ?? null }, + { predicate: "department", value_text: b.department ?? null }, + { predicate: "role_slug", value_text: b.role_slug ?? null }, + ]; + await insertFactsBatch(env, entityId, patches, source, "scrape"); + await Promise.all([ + chan(env, entityId, "email", b.email, source, true), + chan(env, entityId, "linkedin", b.linkedin_url, source, true), + chan(env, entityId, "twitter", b.twitter_url, source), + chan(env, entityId, "phone", b.phone, source), + ]); + if (b.role_slug) + await addTag(env, { entity_id: entityId, taxonomy: "role", slug: b.role_slug, source }); + await enqueueSummaryRebuild(env, entityId); + return entityId; + } + catch (e) { + console.warn("syncBuyerToEntity failed", b.id, e.message); + return null; + } +} +function numOrNull(v) { + return typeof v === "number" && Number.isFinite(v) ? v : null; +} diff --git a/apps/worker/test-dist-q/entities/facts.js b/apps/worker/test-dist-q/entities/facts.js new file mode 100644 index 00000000..a40ff5df --- /dev/null +++ b/apps/worker/test-dist-q/entities/facts.js @@ -0,0 +1,282 @@ +// Fact insertion with content-addressed dedup. The DB trigger +// `trg_facts_supersede` flips the prior is_current=1 fact for the same +// (entity, predicate, source) to 0 after insert. +import { sha256 } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +// Task #8: triggers debounced persona ↔ entity re-match when a fact +// that materially affects scoring is written. No-op otherwise. +import { triggerEntityMatchRefresh, isRelevantPredicate } from "../services/personaMatchTrigger"; +// Task #51: route previously-swallowed fact-write-path failures through the +// structured error logger so they land in `error_log` instead of vanishing. +import { logError } from "../db/error_log"; +export async function insertFact(env, f) { + if (!f.entity_id || !f.predicate) + return null; + const valueKey = JSON.stringify({ + t: f.value_text ?? null, + n: f.value_number ?? null, + j: f.value_json ?? null, + e: f.value_entity_id ?? null, + }); + const hash = await sha256(`${f.entity_id}|${f.predicate}|${valueKey}|${f.source ?? ""}`); + const id = crypto.randomUUID(); + const now = f.observed_at ?? new Date().toISOString(); + // Task #3 (Editable Profiles): lock check. If a locked override exists + // for this (entity, predicate), the new fact row is still inserted (so + // the diff strip can show the AI/scrape attempt) but stamped with + // superseded_by_override=1 so it never wins the read race. The override + // layer overlays at read time via getEffectiveFacts. + // Wrapped in try/catch (not just `.catch`) so a missing `field_overrides` + // table — fresh install ahead of migration 376, or a minimal test DB whose + // prepare() throws synchronously — is logged and degrades to "no lock" + // instead of failing the fact write. + const lock = await (async () => { + try { + return await env.DB.prepare(`SELECT 1 FROM field_overrides + WHERE entity_id = ? AND predicate = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + LIMIT 1`).bind(f.entity_id, f.predicate).first(); + } + catch (e) { + await logError(env, { err: e, step: "facts.insertFact.override_lock_check" }); + return null; + } + })(); + const supersededByOverride = lock ? 1 : 0; + try { + await env.DB.prepare(`INSERT INTO facts ( + id, entity_id, predicate, value_text, value_number, value_json, + value_entity_id, source_kind, source, evidence_url, confidence, + observed_at, valid_from, valid_to, is_current, hash, superseded_by_override + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 1, ?, ?)`).bind(id, f.entity_id, f.predicate, f.value_text ?? null, f.value_number ?? null, f.value_json != null ? JSON.stringify(f.value_json) : null, f.value_entity_id ?? null, f.source_kind, f.source ?? null, f.evidence_url ?? null, f.confidence ?? 1, now, f.valid_from ?? null, f.valid_to ?? null, hash, supersededByOverride).run(); + // Task #3 race fix: the SELECT lock-check above and this INSERT are + // not atomic. If an override landed between them, our row would have + // superseded_by_override=0 even though an override now dominates. The + // override-create handler ALSO runs `UPDATE facts SET + // superseded_by_override = 1` to catch facts inserted before the + // override; this post-insert re-check covers the reverse direction, + // so both writers converge on the same end state regardless of which + // raced first. + if (!supersededByOverride) { + await env.DB.prepare(`UPDATE facts SET superseded_by_override = 1 + WHERE id = ? + AND EXISTS ( + SELECT 1 FROM field_overrides + WHERE entity_id = ? AND predicate = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + )`).bind(id, f.entity_id, f.predicate).run().catch((e) => logError(env, { err: e, step: "facts.insertFact.override_recheck" })); + } + // Centralized rebuild guarantee: every successful fact insert + // enqueues a summary rebuild for the owning entity. This keeps the + // "fact INSERT → rebuild within ~5s" SLO honest regardless of which + // caller wrote the fact (dual-write, merge, manual admin, etc.). + await enqueueSummaryRebuild(env, f.entity_id); + if (isRelevantPredicate(f.predicate)) { + // Fire-and-forget; debounced via KV inside the trigger. + void triggerEntityMatchRefresh(env, f.entity_id).catch((e) => { + console.warn("triggerEntityMatchRefresh from insertFact failed", e.message); + }); + } + // Task #4 (Relationship Inference Worker): debounced enqueue into + // relationship_infer_queue (migration 377). KV-debounced 60s; the + // consolidated nightly slot drains the queue with the per-entity + // orchestrator pass. Never inline — entity/fact writes stay fast. + // No `relationship_infer` JobKind exists, so we fall back to the + // nightly tick per the spec's explicit instruction. + try { + const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); + void enqueueRelInfer(env, f.entity_id, `fact:${f.predicate}`).catch((e) => logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer" })); + } + catch (e) { + void logError(env, { err: e, step: "facts.insertFact.enqueueRelInfer_import" }); + } + return id; + } + catch (e) { + // UNIQUE(hash) collision = exact-replay observation. Task #1 + // requires re-imports of the same Folk row to refresh `observed_at` + // so freshness queries reflect when we last *saw* the fact, even + // when nothing about the value changed. We update the existing row + // (matched by hash) instead of writing a new one. + const msg = e.message || ""; + if (/UNIQUE/i.test(msg)) { + try { + await env.DB.prepare("UPDATE facts SET observed_at = ? WHERE hash = ?").bind(now, hash).run(); + } + catch (uErr) { + console.warn("insertFact observed_at refresh failed", uErr.message); + } + return null; + } + throw e; + } +} +function parseJsonSafe(s) { + if (s == null) + return null; + try { + return JSON.parse(s); + } + catch { + return s; + } +} +export async function loadCurrentOverrides(env, entityId) { + const r = await env.DB.prepare(`SELECT id, predicate, value_text, value_numeric, value_json, overridden_at + FROM field_overrides + WHERE entity_id = ? AND locked = 1 + AND (unlock_after IS NULL OR unlock_after > datetime('now')) + ORDER BY overridden_at DESC`).bind(entityId).all().catch(() => ({ results: [] })); + const map = new Map(); + for (const o of r.results ?? []) { + if (!map.has(o.predicate)) + map.set(o.predicate, o); + } + return map; +} +export async function getEffectiveFacts(env, entityId, opts) { + const factWhere = opts?.includeNonCurrent ? "" : " AND is_current = 1"; + const limit = opts?.limit ?? 500; + const [factsRes, overrides] = await Promise.all([ + env.DB.prepare(`SELECT id, predicate, value_text, value_number, value_json, value_entity_id, + source_kind, source, confidence, verified_score, observed_at, + is_current, superseded_by_override + FROM facts + WHERE entity_id = ?${factWhere} + ORDER BY observed_at DESC LIMIT ?`).bind(entityId, limit).all(), + loadCurrentOverrides(env, entityId), + ]); + const out = []; + const overridePredsSeen = new Set(); + for (const f of factsRes.results ?? []) { + const ov = overrides.get(f.predicate); + if (ov) { + if (!overridePredsSeen.has(f.predicate)) { + overridePredsSeen.add(f.predicate); + out.push({ + id: `override:${ov.id}`, + predicate: f.predicate, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: parseJsonSafe(ov.value_json), + value_entity_id: null, + source_kind: "manual", + source: "field_override", + confidence: 1, + verified_score: null, + observed_at: ov.overridden_at, + is_current: 1, + superseded_by_override: 0, + is_override: true, + override_id: ov.id, + overridden_attempt: false, + }); + } + // Mark the underlying fact as an overridden attempt for the diff + // strip. Never returned as canonical. + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 1, + is_override: false, + override_id: null, + overridden_attempt: true, + }); + } + else if (f.superseded_by_override === 1) { + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 1, + is_override: false, + override_id: null, + overridden_attempt: true, + }); + } + else { + out.push({ + id: f.id, + predicate: f.predicate, + value_text: f.value_text, + value_number: f.value_number, + value_json: parseJsonSafe(f.value_json), + value_entity_id: f.value_entity_id, + source_kind: f.source_kind, + source: f.source, + confidence: f.confidence, + verified_score: f.verified_score, + observed_at: f.observed_at, + is_current: f.is_current, + superseded_by_override: 0, + is_override: false, + override_id: null, + overridden_attempt: false, + }); + } + } + // Overrides for predicates with no underlying fact row at all. + for (const [pred, ov] of overrides.entries()) { + if (overridePredsSeen.has(pred)) + continue; + out.push({ + id: `override:${ov.id}`, + predicate: pred, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: parseJsonSafe(ov.value_json), + value_entity_id: null, + source_kind: "manual", + source: "field_override", + confidence: 1, + verified_score: null, + observed_at: ov.overridden_at, + is_current: 1, + superseded_by_override: 0, + is_override: true, + override_id: ov.id, + overridden_attempt: false, + }); + } + return out; +} +export async function insertFactsBatch(env, entityId, patches, source, sourceKind = "scrape", evidenceUrl = null) { + let n = 0; + for (const p of patches) { + if (p.value_text == null && p.value_number == null && p.value_json == null && p.value_entity_id == null) + continue; + const id = await insertFact(env, { + entity_id: entityId, + predicate: p.predicate, + value_text: p.value_text ?? null, + value_number: p.value_number ?? null, + value_json: p.value_json, + value_entity_id: p.value_entity_id ?? null, + source_kind: sourceKind, + source, + evidence_url: evidenceUrl, + }); + if (id) + n += 1; + } + return n; +} diff --git a/apps/worker/test-dist-q/entities/garbage.js b/apps/worker/test-dist-q/entities/garbage.js new file mode 100644 index 00000000..ffa95393 --- /dev/null +++ b/apps/worker/test-dist-q/entities/garbage.js @@ -0,0 +1,652 @@ +// Task #9: Garbage Entity Detector & Cleanup. +// +// Pure detector (`isGarbage`) flags HTML page titles / nav fragments / +// UI strings polluting `u_entities`. Used by: +// * The pre-insert guard in `createEntity` (rejects before write). +// * The cron sweep (`runCleanupSweep`) that soft-deletes recently- +// created garbage and, on `mode='all'`, performs the one-off pass. +// * The /ops/garbage-review/ console (admin restore / purge). +// +// HONEST DEGRADATION (Task #14 pattern): the optional Workers AI second +// opinion (`aiSecondOpinion`) returns `uncertain` when the `env.AI` +// binding is absent, on any HTTP/network error, or when the JSON +// response is malformed. `evaluateEntity` then DOES NOT flag the +// entity — never silently garbage. +const NAME_MAX_LEN = 80; +// Curated UI / nav strings observed in production on the Investors page. +// Lowercased for comparison. +const KNOWN_UI_STRINGS = new Set([ + "contact us", "contact", "search icon", "search", "home", "about", + "about us", "menu", "login", "log in", "sign in", "sign up", + "sign-up", "register", "our team", "team", "limited partners", + "portfolio", "our portfolio", "careers", "jobs", "privacy", + "privacy policy", "terms", "terms of service", "cookies", + "cookie policy", "blog", "news", "press", "press releases", + "get in touch", "subscribe", "newsletter", "footer", "header", + "navigation", "nav", "skip to content", "back to top", "read more", + "learn more", "view all", "see all", "all rights reserved", + "follow us", "share", "tweet", "facebook", "twitter", "linkedin", + "instagram", "youtube", "the team", "our story", +]); +// Heuristic leaders that strongly indicate a press/blog post title +// got captured as an "entity". +const LEADER_RE = /^(introducing|announcing|welcome to|how|why|what|when|where|the future of|inside)\s+/i; +// Page-title with `|`-separated domain/brand fragment. Examples: +// "Our Team | Sequoia Capital", "Contact Tenity | Get in Touch", +// "Home | Sequoia Capital". +const PIPE_TITLE_RE = /\s\|\s\S/; +// Pure emoji / icon names (no alphanumerics at all). +const NO_ALNUM_RE = /^[^\p{L}\p{N}]+$/u; +// Listicle / directory page titles captured as entity names. +// +// This is the gap that let ~128 non-firms into the firms table: a crawler +// ingested aggregator pages ("VC Firms By Stage" on failory.com) and made +// one entity per outbound link, taking the page title as the name. The +// existing rules could not catch it — such a title has no pipe fragment, no +// press leader, is well under 80 characters and is not a known nav string. +// +// Deliberately narrow, because a single matched reason marks an entity +// garbage. Each pattern is a phrase a real firm name essentially never +// contains: "Top Tier Capital Partners" and "Stage Fund" are real firms, so +// a bare leading "top" or the word "stage" alone must NOT match. +const LISTICLE_RES = [ + // "VC Firms By Stage", "Investors by sector", "Funds per geography" + /\b(?:by|per)\s+(?:stage|sector|industry|geograph|countr|region|check\s*size|vertical)/i, + // "Top 50 VC Firms", "Best 10 Seed Funds" — the number is what makes this + // safe; a leading "Top"/"Best" alone is a legitimate name fragment. + /^(?:the\s+)?(?:top|best|leading)\s+\d+\b/i, + // "List of European VCs", "Directory of angel investors" + /\b(?:list|directory|database|roundup|ranking)\s+of\s+/i, + // "The Ultimate Guide to Seed Funds", "Complete List of ..." + /^(?:the\s+)?(?:complete|ultimate|definitive)\s+(?:list|guide|directory|database)\b/i, +]; +// Minimum length before a name that is identical to its own domain slug is +// treated as URL-derived rather than a genuine single-word brand. +// "Stripe" / "Coatue" / "Atomico" are real names that equal their domain; +// "Firstmarkcap" (firstmarkcap.com) and "Forerunnerventures" are slugs that +// were title-cased because no real name was ever extracted. +const SLUG_NAME_MIN_LEN = 12; +/** + * True when the display name is just the registrable domain label with the + * first letter capitalised — i.e. the crawler never found a name and fell + * back to the URL. Length-gated so short single-word brands are untouched. + */ +function looksDomainDerived(name, domain) { + if (!domain) + return false; + const label = domain.toLowerCase().replace(/^www\./, "").split(".")[0] ?? ""; + if (label.length < SLUG_NAME_MIN_LEN) + return false; + const n = name.trim().toLowerCase(); + // A genuine name carries separators the slug cannot ("First Mark Capital"). + if (/[\s.\-_]/.test(n)) + return false; + return n === label; +} +// --------------------------------------------------------------------------- +// Task #6: person-name disambiguation. Classifies a name that was recorded +// as a `person` into one of: a real person, an organization scraped as a +// person (firm / fund / accelerator / company), generic page junk, or +// uncertain. Used by: +// * the pre-insert reclassify-on-write guard in `createEntity`, +// * the cron / one-off sweep (`runCleanupSweep`), +// * the scraper extraction boundary (`extractPeopleFromPage`). +// PURE — name-only, no IO — so it's safe on the hot write path. +// --------------------------------------------------------------------------- +// Legal-entity suffixes — an extremely strong organization signal anywhere +// in the name. Normalized (punctuation stripped) before comparison. +const ORG_LEGAL_SUFFIX = new Set([ + "llc", "inc", "ltd", "limited", "lp", "llp", "plc", "gmbh", "ag", + "sarl", "bv", "pty", "oy", "ab", "srl", "spa", +]); +// Descriptor words that, as the LAST token, denote an organization +// ("Intel Capital", "Mendoza Ventures", "Hillman Accelerator Foundation"). +const ORG_SUFFIX_LAST = new Set([ + "capital", "ventures", "venture", "partners", "partner", "holdings", + "group", "fund", "funds", "foundation", "labs", "lab", "hub", + "collective", "management", "advisors", "associates", "accelerator", + "incubator", "equity", "securities", "technologies", "studios", "studio", + "network", "institute", "academy", "council", "alliance", "syndicate", + "consortium", "enterprises", "industries", "international", "global", + "company", "corp", "corporation", "university", "college", "systems", + "solutions", +]); +// Generic, non-distinctive words. A name made up ENTIRELY of these is junk +// ("Deep Tech", "Our Mission"); they're also excluded when looking for a +// distinctive proper-noun token in an org name. +const GENERIC_WORDS = new Set([ + "the", "our", "your", "my", "a", "an", "all", "more", "new", "updated", + "featured", "latest", "recent", "top", "best", "of", "and", "or", "for", + "with", "to", "in", "on", "at", "by", "from", "about", "welcome", "hello", + "home", "homepage", "page", "web", "website", "webpage", "mission", + "vision", "values", "story", "team", "careers", "jobs", "blog", "news", + "press", "media", "map", "menu", "footer", "header", "sidebar", "gallery", + "resources", "events", "podcast", "newsletter", "insights", "research", + "report", "reports", "overview", "summary", "services", "solutions", + "products", "pricing", "features", "get", "started", "learn", "read", + "view", "see", "guide", "guides", "faq", "faqs", "help", "support", + "contact", "deep", "tech", "technology", "startup", "startups", + "mentorship", "money", "data", "signal", "community", "ecosystem", + "platform", "world", "global", "international", "region", "regions", + "area", "areas", "north", "south", "east", "west", "central", "america", + "americas", "europe", "asia", "africa", "oceania", "antarctica", "middle", + "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", + "thumbnail", "placeholder", "avatar", "headshot", "slideshow", "carousel", + "machine", "wayback", "future", "work", "working", "people", "portfolio", + "companies", "investors", "founders", "funding", "rounds", "deals", +]); +// Words whose presence alone marks a name as page junk rather than a +// person — decorative / UI / asset captions that pass NAME_RE. +const HARD_JUNK_WORDS = new Set([ + "image", "images", "photo", "photos", "logo", "logos", "icon", "banner", + "thumbnail", "placeholder", "gallery", "slideshow", "carousel", "homepage", + "webpage", "sidebar", "footer", "header", "menu", "map", "wayback", + "machine", +]); +// Exact lowercase phrases observed polluting the People list. +const KNOWN_JUNK_PHRASES = new Set([ + "updated homepage image", "our mission", "map of the money", + "wayback machine", "deep tech", "startup mentorship hub", "read more", + "learn more", "our team", "the team", "our story", "our values", + "our vision", "get started", "coming soon", "page not found", +]); +// Exact lowercase place / region names that get scraped as "people". +const PLACE_NAMES = new Set([ + "north america", "south america", "central america", "latin america", + "united states", "united kingdom", "middle east", "european union", + "north", "south", "east", "west", "europe", "asia", "africa", "oceania", + "antarctica", "americas", "global", "worldwide", +]); +function normToken(t) { + return t.toLowerCase().normalize("NFKD").replace(/[^a-z0-9]/g, ""); +} +function orgRoleForTokens(tokensLower) { + const set = new Set(tokensLower); + if (set.has("accelerator") || set.has("incubator")) + return "accelerator"; + if (set.has("fund") || set.has("funds")) + return "fund"; + for (const t of ["capital", "ventures", "venture", "partners", "partner", + "equity", "management", "advisors", "associates", "holdings", + "securities", "syndicate"]) { + if (set.has(t)) + return "investor_firm"; + } + return "firm"; +} +/** Map an inferred org role to the `firms.kind` taxonomy for dual-write. */ +export function orgRoleToFirmKind(role) { + switch (role) { + case "accelerator": return "accelerator"; + case "fund": return "fund"; + case "investor_firm": return "vc"; + default: return null; + } +} +/** + * Pure classifier for a name recorded as a `person`. Conservative by + * design: only returns `organization` / `junk` when the signal is clear, + * otherwise `person` (a plausible human name) or `uncertain`. Callers + * decide what to DO with each verdict (reclassify, soft-delete, review). + */ +export function classifyPersonName(rawName) { + const raw = (rawName ?? "").trim(); + if (!raw) + return { verdict: "junk", reasons: ["empty_name"] }; + const lower = raw.toLowerCase(); + const tokens = raw.split(/\s+/).filter(Boolean); + const norm = tokens.map(normToken).filter(Boolean); + // 1. Exact known-junk phrase / place name. + if (KNOWN_JUNK_PHRASES.has(lower)) + return { verdict: "junk", reasons: ["known_junk_phrase"] }; + if (PLACE_NAMES.has(lower)) + return { verdict: "junk", reasons: ["place_name"] }; + // 2. Hard junk word present (image / logo / homepage / map / ...). + // PRECISION GUARD (precision-over-recall): a single junk token inside an + // otherwise-clean two-token Title-Case name ("John Banner", "John Map") + // must NOT auto-delete a plausible real person. Real decorative captions + // are ≥3 tokens ("Updated Homepage Image") or all-generic two-token + // phrases ("Wayback Machine", caught by rule 4 below), so deferring the + // junk-word rule for clean two-token names keeps every junk fixture while + // protecting people whose surname happens to collide with an asset word. + const cleanTwoToken = tokens.length === 2 && tokens.every((t) => /^[\p{Lu}][\p{L}'’.\-]*$/u.test(t)); + if (!cleanTwoToken) { + for (const t of norm) { + if (HARD_JUNK_WORDS.has(t)) + return { verdict: "junk", reasons: [`junk_word:${t}`] }; + } + } + // 3. Organization-suffix detection. + const last = norm[norm.length - 1] ?? ""; + const hasLegal = norm.some((t) => ORG_LEGAL_SUFFIX.has(t)); + const lastIsOrgSuffix = ORG_SUFFIX_LAST.has(last); + if (hasLegal || lastIsOrgSuffix) { + // Need a distinctive (non-generic, non-suffix) token to call it a real + // org. "Intel Capital" → distinctive "intel". "Startup Mentorship Hub" + // → all-generic + suffix → junk. + const distinctive = norm.filter((t, i) => !GENERIC_WORDS.has(t) && + !ORG_SUFFIX_LAST.has(t) && + !ORG_LEGAL_SUFFIX.has(t) && + !(i === norm.length - 1 && lastIsOrgSuffix)); + if (distinctive.length === 0) { + return { verdict: "junk", reasons: ["generic_org_phrase"] }; + } + return { + verdict: "organization", + orgRole: orgRoleForTokens(norm), + reasons: [hasLegal ? "org_legal_suffix" : `org_suffix:${last}`], + }; + } + // 4. Every token is a generic word ("Deep Tech", "Our Mission"). + if (norm.length >= 1 && norm.every((t) => GENERIC_WORDS.has(t))) { + return { verdict: "junk", reasons: ["all_generic_words"] }; + } + // 5. Plausible human name: 2–4 tokens, ≥2 capitalized, not all generic. + const titleTokens = tokens.filter((t) => /^[\p{Lu}]/u.test(t)); + if (tokens.length >= 2 && tokens.length <= 4 && titleTokens.length >= 2) { + return { verdict: "person", reasons: ["plausible_person_name"] }; + } + // 6. Anything else — don't guess. + return { verdict: "uncertain", reasons: ["unclassified"] }; +} +/** Pure detector. NO IO. Safe to call inline on every entity write. */ +export function isGarbage(input) { + const reasons = []; + const raw = (input.display_name ?? "").trim(); + // Rule 1: empty or whitespace-only name. + if (!raw) { + reasons.push("empty_name"); + return { is_garbage: true, reasons }; + } + // Rule 2: name longer than 80 chars. + if (raw.length > NAME_MAX_LEN) + reasons.push("name_too_long"); + // Rule 3: pure emoji / icon (no letters or digits). + if (NO_ALNUM_RE.test(raw)) + reasons.push("no_alphanumeric_chars"); + // Rule 4: page-title with `|` brand fragment. + if (PIPE_TITLE_RE.test(raw)) + reasons.push("page_title_pipe_fragment"); + // Rule 5: blog/press leader phrase. + if (LEADER_RE.test(raw)) + reasons.push("press_leader_phrase"); + // Rule 6: known UI / nav string (case-insensitive exact match). + if (KNOWN_UI_STRINGS.has(raw.toLowerCase())) + reasons.push("known_ui_string"); + // Rule 6c: listicle / directory page title captured as an entity name. + if (LISTICLE_RES.some((re) => re.test(raw))) + reasons.push("listicle_page_title"); + // Rule 6d: name is just the domain slug — the crawler never extracted a + // real name and fell back to the URL. + if (looksDomainDerived(raw, input.primary_domain)) + reasons.push("domain_slug_name"); + // Rule 6b (Task #6 Section A/F): literal HTML entity in name + // (e.g. "Founder & Partner", "Acme & Co"). These are parser + // bugs upstream — the entity should have been decoded before write. + // We flag them here as garbage so they soft-delete on the next sweep + // and surface to the operator console; the durable fix is at the + // scraper layer via decodeEntities() (Task #6 Section F). + if (/&(amp|lt|gt|quot|#x?[0-9a-f]+);/i.test(raw)) + reasons.push("literal_html_entity"); + // Rule 7: person-specific constraints — must contain a space AND + // must not contain pipe / slash / colon. Real human display names + // are "First Last", not "Contact | Sequoia" or "team/people:1". + if (input.kind === "person") { + if (!/\s/.test(raw)) + reasons.push("person_no_space"); + if (/[|/:]/.test(raw)) + reasons.push("person_contains_separator"); + // Task #6: generic page-junk names recorded as people ("Updated + // Homepage Image", "Our Mission", "North America"). Organization + // names are NOT flagged here — they're reclassified (not deleted) + // by the createEntity write guard and the sweep. + const cls = classifyPersonName(raw); + if (cls.verdict === "junk") { + for (const code of cls.reasons) + reasons.push(`name_${code}`); + } + } + return { is_garbage: reasons.length > 0, reasons }; +} +// --------------------------------------------------------------------------- +// Structural rule (requires DB lookups): zero facts AND zero relationships +// AND zero contact channels AND crawler-created >24h ago. Used by the +// cron sweep — NOT by the pre-insert guard (the entity hasn't been +// written yet, so it has no joins). +// --------------------------------------------------------------------------- +export async function isStructurallyOrphan(env, entityId, options = {}) { + const minAge = options.minAgeHours ?? 24; + const reasons = []; + try { + const row = await env.DB.prepare(`SELECT + (SELECT COUNT(*) FROM facts WHERE entity_id = ?1) AS facts, + (SELECT COUNT(*) FROM rel_edges WHERE src_entity_id = ?1 OR dst_entity_id = ?1) AS rels, + (SELECT COUNT(*) FROM channels WHERE entity_id = ?1) AS chans, + (SELECT (julianday('now') - julianday(created_at)) * 24 FROM u_entities WHERE id = ?1) AS age_hours`).bind(entityId).first(); + if (!row) + return { orphan: false, reasons }; + const ageHours = Number(row.age_hours ?? 0); + if (Number(row.facts) === 0 && Number(row.rels) === 0 && Number(row.chans) === 0 && ageHours >= minAge) { + reasons.push("structural_orphan_no_signal"); + return { orphan: true, reasons }; + } + } + catch (e) { + // Optional source tables (channels) may be missing in test + // DBs — degrade to "not orphan" rather than throwing. Per the + // Task #14 honest-degradation pattern. + console.warn("isStructurallyOrphan probe failed", entityId, e.message); + } + return { orphan: false, reasons }; +} +const AI_PROMPT = `You are a data-quality filter for a CRM. Given a candidate \ +entity record, decide whether the display_name is a real person/organization \ +name or noise scraped from an HTML page (page titles, nav labels, "Contact Us", \ +press headlines like "Introducing X", marketing blurbs, etc.). +Reply ONLY as compact JSON: {"verdict":"garbage|real|uncertain","confidence":0.0-1.0,"reason":""}.`; +export async function aiSecondOpinion(env, input) { + if (!env.AI || typeof env.AI.run !== "function") { + return { verdict: "uncertain", confidence: 0, reason: "ai_binding_missing" }; + } + const payload = { + kind: input.kind, + display_name: input.display_name ?? null, + primary_url: input.primary_url ?? null, + primary_domain: input.primary_domain ?? null, + }; + try { + const res = (await env.AI.run("@cf/meta/llama-3.1-8b-instruct-fast", { + messages: [ + { role: "system", content: AI_PROMPT }, + { role: "user", content: JSON.stringify(payload) }, + ], + max_tokens: 80, + })); + const text = typeof res === "string" ? res : (res?.response ?? ""); + const match = text.match(/\{[\s\S]*\}/); + if (!match) + return { verdict: "uncertain", confidence: 0, reason: "ai_no_json" }; + const parsed = JSON.parse(match[0]); + const verdict = parsed.verdict === "garbage" || parsed.verdict === "real" ? parsed.verdict : "uncertain"; + const conf = typeof parsed.confidence === "number" && Number.isFinite(parsed.confidence) + ? Math.max(0, Math.min(1, parsed.confidence)) + : 0; + return { verdict, confidence: conf, reason: typeof parsed.reason === "string" ? parsed.reason : undefined }; + } + catch (e) { + return { verdict: "uncertain", confidence: 0, reason: "ai_error:" + e.message }; + } +} +/** + * Combined verdict: heuristic detector + optional AI second opinion + * for names 30–60 chars that don't match any heuristic. AI flags only + * when verdict='garbage' AND confidence > 0.8. When AI is unavailable + * or returns 'uncertain', the entity is NOT flagged. + */ +export async function evaluateEntity(env, input, opts = {}) { + const heur = isGarbage(input); + if (heur.is_garbage) + return heur; + if (opts.skipAi) + return heur; + const name = (input.display_name ?? "").trim(); + if (name.length < 30 || name.length > 60) + return heur; + const ai = await aiSecondOpinion(env, input); + if (ai.verdict === "garbage" && ai.confidence > 0.8) { + return { is_garbage: true, reasons: ["ai_second_opinion", `ai_conf:${ai.confidence.toFixed(2)}`] }; + } + return heur; +} +// --------------------------------------------------------------------------- +// Soft-delete / restore / purge helpers. All write through +// `data_quality_log` so the operator console can audit every transition. +// --------------------------------------------------------------------------- +export async function logDataQuality(env, entityId, issue, reasons, source, actorEmail, +/** Stored in reasons_json instead of `reasons` when given. Used to park a + * structured snapshot (see softDeleteEntity) rather than a reason list. */ +payload) { + try { + await env.DB.prepare(`INSERT INTO data_quality_log (entity_id, issue, reasons_json, source, actor_email) + VALUES (?, ?, ?, ?, ?)`).bind(entityId, issue, JSON.stringify(payload ?? reasons), source, actorEmail ?? null).run(); + } + catch (e) { + console.warn("data_quality_log insert failed", entityId, e.message); + } +} +export async function softDeleteEntity(env, entityId, reasons, source, actorEmail) { + await env.DB.prepare(`UPDATE u_entities + SET status = 'soft_deleted', + deleted_reason = COALESCE(deleted_reason, ?), + updated_at = datetime('now') + WHERE id = ? AND status != 'soft_deleted'`).bind("garbage_detector_v1:" + reasons.join(","), entityId).run(); + try { + // Park the roles before deleting them. Without this the delete is a + // one-way door: restoreEntity flips status back to 'active' and nothing + // puts the roles back, so a restored entity returns with no roles at all + // and stays invisible on every role-filtered surface — the investor + // lists, the persona matchers, the founder screens. That makes + // "quarantine, then restore the false positives" a promise the code + // cannot keep, which is the whole point of soft delete over purge. + const roles = await env.DB.prepare(`SELECT role, is_primary, source, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(); + // Parked unconditionally, including the empty case. If the park were + // written only when roles exist, an entity that was restored and later + // soft-deleted again with no roles would leave the first park as the + // newest one, and the second restore would replay roles the entity no + // longer had. One row per soft-delete keeps "newest park" and "newest + // soft-delete" the same event. + await logDataQuality(env, entityId, "soft_deleted_roles", [], source, actorEmail, roles.results ?? []); + await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entityId).run(); + } + catch (e) { + console.warn("entity_roles delete during soft-delete failed", entityId, e.message); + } + await logDataQuality(env, entityId, "soft_deleted", reasons, source, actorEmail); +} +export async function restoreEntity(env, entityId, actorEmail) { + await env.DB.prepare(`UPDATE u_entities + SET status = 'active', + deleted_reason = NULL, + updated_at = datetime('now') + WHERE id = ?`).bind(entityId).run(); + // Put back the roles the soft delete removed. Restoring status alone + // returned an entity with no roles, so it never reappeared on any + // role-filtered surface — the restore looked like it worked and did not. + let rolesRestored = 0; + try { + const parked = await env.DB.prepare(`SELECT reasons_json FROM data_quality_log + WHERE entity_id = ? AND issue = 'soft_deleted_roles' + ORDER BY detected_at DESC, id DESC LIMIT 1`).bind(entityId).first(); + const rows = parked?.reasons_json ? JSON.parse(parked.reasons_json) : []; + for (const r of Array.isArray(rows) ? rows : []) { + if (!r || typeof r.role !== "string" || !r.role) + continue; + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, role) DO NOTHING`).bind(entityId, r.role, r.is_primary ?? 0, r.source ?? "restore", r.confidence ?? 0.5).run(); + rolesRestored += 1; + } + } + catch (e) { + // A restore that cannot replay the roles is still better than none — + // but it must not look clean, so it is recorded rather than swallowed. + await logDataQuality(env, entityId, "restore_roles_failed", [e.message], "operator", actorEmail); + } + await logDataQuality(env, entityId, "restored", [`roles_restored:${rolesRestored}`], "operator", actorEmail); +} +export async function purgeEntity(env, entityId, actorEmail) { + // Best-effort cascade across the optional referencing tables; each + // wrapped in its own try/catch so a missing table doesn't block the + // primary delete. Per the Task #14 honest-degradation pattern. + const cascades = [ + `DELETE FROM facts WHERE entity_id = ?`, + `DELETE FROM rel_edges WHERE src_entity_id = ? OR dst_entity_id = ?`, + `DELETE FROM channels WHERE entity_id = ?`, + `DELETE FROM entity_roles WHERE entity_id = ?`, + `DELETE FROM entity_history WHERE entity_id = ?`, + `DELETE FROM entity_legacy_map WHERE entity_id = ?`, + ]; + for (const sql of cascades) { + try { + if (sql.includes("OR dst_entity_id")) { + await env.DB.prepare(sql).bind(entityId, entityId).run(); + } + else { + await env.DB.prepare(sql).bind(entityId).run(); + } + } + catch (e) { + // table-missing or FK noise — log and continue + console.warn("purge cascade failed", sql.slice(0, 40), e.message); + } + } + // Log BEFORE the final delete so the audit trail survives even if + // the row-delete races a concurrent reader. data_quality_log keeps + // entity_id as TEXT (no FK), so the row remains queryable. + await logDataQuality(env, entityId, "purged", [], "operator", actorEmail); + await env.DB.prepare(`DELETE FROM u_entities WHERE id = ?`).bind(entityId).run(); +} +// --------------------------------------------------------------------------- +// Task #6: reclassify a person row that is actually an organization. The +// row is FLIPPED in place (kind person→org) so it leaves the People list +// and joins the org world — non-destructive and reversible (the row, its +// facts and relationships are preserved). When a domain/website is known +// we dual-write a `firms` row so it also surfaces in the Firms list; +// upsertFirm's syncFirmToEntity re-resolves the firm to THIS now-org +// entity via the primary_domain match, so no duplicate entity is minted. +// HONEST DEGRADATION: with no domain/website we cannot dedupe a firm row +// (name-only matching mints duplicates), so we skip it and record the gap +// rather than guessing — the entity still leaves People as an org. +// --------------------------------------------------------------------------- +export async function reclassifyPersonAsOrg(env, entity, orgRole, reasons, source, actorEmail) { + // 1. Flip kind in place — removes it from the People list immediately. + await env.DB.prepare(`UPDATE u_entities SET kind = 'org', updated_at = datetime('now') WHERE id = ?`).bind(entity.id).run(); + // 2. Swap person/investor roles for the inferred org role. Capture the + // prior role set into the audit trail FIRST so the reclassification is + // fully reversible: an operator (or a rollback) can restore the original + // roles from the data_quality_log row, not just flip the kind back. + let priorRoles = []; + try { + const existing = await env.DB.prepare(`SELECT role FROM entity_roles WHERE entity_id = ?`).bind(entity.id).all(); + priorRoles = (existing.results ?? []).map((x) => x.role); + await env.DB.prepare(`DELETE FROM entity_roles WHERE entity_id = ?`).bind(entity.id).run(); + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, 1, ?, 1) + ON CONFLICT(entity_id, role) DO UPDATE SET is_primary = 1`).bind(entity.id, orgRole, source).run(); + } + catch (e) { + console.warn("reclassify role swap failed", entity.id, e.message); + } + // 3. Dual-write a firms row when we can dedupe by domain/website. + let firmListed = false; + if (entity.display_name && (entity.primary_domain || entity.primary_url)) { + try { + const { upsertFirm } = await import("../scraper/firms_upsert.js"); + await upsertFirm(env, { + name: entity.display_name, + domain: entity.primary_domain ?? null, + website: entity.primary_url ?? null, + kind: orgRoleToFirmKind(orgRole), + }, source); + firmListed = true; + } + catch (e) { + console.warn("reclassify upsertFirm failed", entity.id, e.message); + } + } + await logDataQuality(env, entity.id, "reclassified", [...reasons, `org_role:${orgRole}`, + priorRoles.length ? `prior_roles:${priorRoles.join(",")}` : "prior_roles:none", + firmListed ? "firm_listed" : "firm_row_skipped_no_domain"], source, actorEmail); + return { reclassified: true, firm_listed: firmListed }; +} +export async function runCleanupSweep(env, opts = {}) { + const mode = opts.mode ?? "recent"; + const lookback = opts.lookbackHours ?? 24; + const limit = opts.limit ?? 5000; + const source = opts.source ?? (mode === "all" ? "oneoff_cleanup" : "cron_sweep"); + const where = mode === "all" + ? `status NOT IN ('soft_deleted','merged')` + : `status NOT IN ('soft_deleted','merged') AND created_at >= datetime('now', '-${lookback} hours')`; + const rows = await env.DB.prepare(`SELECT id, kind, display_name, primary_url, primary_domain, primary_email_key, primary_linkedin_key + FROM u_entities + WHERE ${where} + ORDER BY created_at DESC + LIMIT ?`).bind(limit).all(); + const items = rows.results ?? []; + const byReason = {}; + let flagged = 0; + let softDeleted = 0; + let reclassified = 0; + let needsReview = 0; + for (const r of items) { + // Task #6: organization-name disambiguation for `person` rows. + // Orgs scraped as people are RECLASSIFIED into the org world (and the + // Firms list) rather than soft-deleted. A strong personal identifier + // (personal LinkedIn /in/ or an email) contradicting an org-suffix + // name is flagged for operator review instead of auto-flipped — never + // destroy a likely real person. Junk / plausible-person / uncertain + // names fall through to the existing garbage + orphan path below so + // no prior behavior regresses. + if (r.kind === "person") { + const cls = classifyPersonName(r.display_name); + if (cls.verdict === "organization" && cls.orgRole) { + const personalLinkedin = !!r.primary_linkedin_key && /(^|\/)in\//i.test(r.primary_linkedin_key); + const hasEmail = !!r.primary_email_key; + if (personalLinkedin || hasEmail) { + for (const code of cls.reasons) + byReason[`review_${code}`] = (byReason[`review_${code}`] ?? 0) + 1; + await logDataQuality(env, r.id, "needs_review", [...cls.reasons, "org_name_with_person_signal"], source, opts.actorEmail ?? null); + needsReview += 1; + continue; + } + try { + await reclassifyPersonAsOrg(env, { id: r.id, display_name: r.display_name, primary_url: r.primary_url, primary_domain: r.primary_domain }, cls.orgRole, cls.reasons, source, opts.actorEmail ?? null); + reclassified += 1; + byReason[`reclassified_${cls.orgRole}`] = (byReason[`reclassified_${cls.orgRole}`] ?? 0) + 1; + } + catch (e) { + console.warn("sweep reclassify failed", r.id, e.message); + } + continue; + } + } + // Route through evaluateEntity so the AI second opinion fires for + // ambiguous 30–60 char names (when env.AI is bound). Honors + // skipAi for unit tests + operator-requested fast sweeps. + const evald = await evaluateEntity(env, { + kind: r.kind, display_name: r.display_name, + primary_url: r.primary_url, primary_domain: r.primary_domain, + primary_email_key: r.primary_email_key, primary_linkedin_key: r.primary_linkedin_key, + }, { skipAi: opts.skipAi }); + let reasons = evald.reasons; + let flag = evald.is_garbage; + if (!flag) { + // Structural rule — only meaningful when the entity is not + // brand-new (otherwise the crawler may still be writing joins). + const orphan = await isStructurallyOrphan(env, r.id, { minAgeHours: 24 }); + if (orphan.orphan) { + flag = true; + reasons = orphan.reasons; + } + } + if (!flag) + continue; + flagged += 1; + for (const code of reasons) + byReason[code] = (byReason[code] ?? 0) + 1; + try { + await softDeleteEntity(env, r.id, reasons, source, opts.actorEmail ?? null); + softDeleted += 1; + } + catch (e) { + console.warn("sweep soft-delete failed", r.id, e.message); + } + } + const result = { + scanned: items.length, flagged, soft_deleted: softDeleted, + reclassified, needs_review: needsReview, + by_reason: byReason, bounded: items.length >= limit, + }; + console.log("garbage.cleanup_sweep", JSON.stringify({ mode, ...result })); + return result; +} diff --git a/apps/worker/test-dist-q/entities/model.js b/apps/worker/test-dist-q/entities/model.js new file mode 100644 index 00000000..a09d5fab --- /dev/null +++ b/apps/worker/test-dist-q/entities/model.js @@ -0,0 +1,6 @@ +// TS types matching the unified entity DDL (migrations 200–208). +// Task #4: re-export the rich-profile write helpers under a single +// EntityService facade so callers can `import { EntityService } from +// "./entities/model"` for every structured profile write. +export { EntityService } from "./profile"; +export { PREDICATE_REGISTRY, PREDICATE_MAP, EMITTED_PREDICATES, getPredicateMeta, } from "./profile-predicates"; diff --git a/apps/worker/test-dist-q/entities/normalize.js b/apps/worker/test-dist-q/entities/normalize.js new file mode 100644 index 00000000..84318cba --- /dev/null +++ b/apps/worker/test-dist-q/entities/normalize.js @@ -0,0 +1,120 @@ +// Canonical key generators for the unified entity model. Channels are +// keyed by (kind, canonical) and equality must collapse trivial +// presentation differences — case, whitespace, '+suffix' in emails, +// formatting characters in phone numbers, query strings on LinkedIn URLs. +export function canonicalEmail(raw) { + if (!raw) + return null; + const s = String(raw).trim().toLowerCase(); + const m = /^([^@\s]+)@([a-z0-9.-]+\.[a-z]{2,})$/i.exec(s); + if (!m) + return null; + // Strip '+suffix' from the local part (Gmail-style). + const local = m[1].split("+")[0]; + return `${local}@${m[2]}`; +} +export function canonicalPhone(raw) { + if (!raw) + return null; + const digits = String(raw).replace(/[^\d+]/g, ""); + if (!digits) + return null; + // Best-effort E.164: keep leading + if present, else assume already + // includes country code. + return digits.startsWith("+") ? digits : `+${digits.replace(/^\++/, "")}`; +} +export function canonicalLinkedin(raw) { + if (!raw) + return null; + let s = String(raw).trim(); + if (!s) + return null; + // Accept handle-only ("janedoe") or full URL. + if (!/^https?:\/\//i.test(s) && !s.includes("/")) { + return `/in/${s.toLowerCase()}`; + } + try { + const u = new URL(s.startsWith("http") ? s : `https://${s}`); + if (!/linkedin\.com$/i.test(u.hostname) && !/(^|\.)linkedin\.com$/i.test(u.hostname)) + return null; + const p = u.pathname.replace(/\/+$/, "").toLowerCase(); + // /in/ | /company/ | /school/ + const m = /^\/(in|company|school|pub)\/([^/?#]+)/.exec(p); + if (!m) + return null; + return `/${m[1] === "pub" ? "in" : m[1]}/${m[2]}`; + } + catch { + return null; + } +} +export function canonicalTwitter(raw) { + if (!raw) + return null; + const s = String(raw).trim().replace(/^@/, ""); + if (!s) + return null; + if (/^https?:\/\//i.test(s)) { + try { + const u = new URL(s); + const m = /^\/([A-Za-z0-9_]{1,15})\/?$/.exec(u.pathname); + return m ? m[1].toLowerCase() : null; + } + catch { + return null; + } + } + return /^[A-Za-z0-9_]{1,15}$/.test(s) ? s.toLowerCase() : null; +} +export function canonicalGithub(raw) { + if (!raw) + return null; + const s = String(raw).trim().replace(/^@/, ""); + if (!s) + return null; + if (/^https?:\/\//i.test(s)) { + try { + const u = new URL(s); + const m = /^\/([A-Za-z0-9-]{1,39})\/?$/.exec(u.pathname); + return m ? m[1].toLowerCase() : null; + } + catch { + return null; + } + } + return /^[A-Za-z0-9-]{1,39}$/.test(s) ? s.toLowerCase() : null; +} +export function canonicalDomain(raw) { + if (!raw) + return null; + let s = String(raw).trim().toLowerCase(); + if (!s) + return null; + if (/^https?:\/\//.test(s)) { + try { + s = new URL(s).hostname; + } + catch { + return null; + } + } + s = s.replace(/^www\./, "").replace(/\/.*$/, ""); + return /^[a-z0-9.-]+\.[a-z]{2,}$/.test(s) ? s : null; +} +export function canonicalUrl(raw) { + if (!raw) + return null; + try { + const u = new URL(String(raw).trim()); + u.hash = ""; + return u.toString(); + } + catch { + return null; + } +} +export async function sha256(s) { + const buf = new TextEncoder().encode(s); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} diff --git a/apps/worker/test-dist-q/entities/profile-predicates.js b/apps/worker/test-dist-q/entities/profile-predicates.js new file mode 100644 index 00000000..f88842dc --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile-predicates.js @@ -0,0 +1,218 @@ +// Task #4: Predicate registry — single source of truth. +// +// The same list is mirrored into `predicate_registry` by migration 328. +// Every predicate string that `profile.ts` ever passes to `insertFact` +// MUST appear in this array; the CI smoke test (test/profile.test.mjs) +// enforces it. Adding a new predicate is a TWO-file change: +// 1. append it here AND +// 2. add the matching `INSERT OR IGNORE INTO predicate_registry` row in +// migration 328_predicate_registry.sql. +// +// `value_type` is informational metadata for the UI formatter — it does +// not constrain the runtime shape of the fact value_json (those shapes +// live in profile-shapes.ts). +export const PREDICATE_REGISTRY = [ + // ---- identity (rich-profile) ------------------------------------------- + { predicate: "person.identity", label: "Identity snapshot", icon: "user", formatter: "json", category: "identity", value_type: "json", description: "Snapshot of person_identity row." }, + { predicate: "person.identity.full_name", label: "Full name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Legal or commonly used full name." }, + { predicate: "person.identity.preferred_name", label: "Preferred name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "What the person prefers to be called." }, + { predicate: "person.identity.pronouns", label: "Pronouns", icon: "user-circle", formatter: "text", category: "identity", value_type: "json", description: "Pronouns triple (subject/object/possessive)." }, + { predicate: "person.identity.birth_year", label: "Birth year", icon: "cake", formatter: "year", category: "identity", value_type: "year", description: "Year of birth (no full DOB stored)." }, + { predicate: "person.identity.nationality", label: "Nationality", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO-3166-1 alpha-2 nationality." }, + { predicate: "person.identity.languages", label: "Languages", icon: "languages", formatter: "list", category: "identity", value_type: "json", description: "Spoken languages with proficiency." }, + { predicate: "person.identity.timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "IANA tz database name." }, + { predicate: "person.identity.location_city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Current city of residence." }, + { predicate: "person.identity.location_country", label: "Country", icon: "globe", formatter: "flag", category: "identity", value_type: "country_iso2", description: "Current country of residence." }, + { predicate: "person.identity.headshot_url", label: "Headshot", icon: "image", formatter: "avatar", category: "identity", value_type: "url", description: "Public headshot image URL." }, + // ---- career ------------------------------------------------------------- + { predicate: "person.career", label: "Career entry", icon: "briefcase", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a career_history row." }, + // ---- board -------------------------------------------------------------- + { predicate: "person.board_seat", label: "Board seat", icon: "users", formatter: "json", category: "career", value_type: "json", description: "Snapshot of a board_seats row." }, + // ---- education ---------------------------------------------------------- + { predicate: "person.education", label: "Education", icon: "graduation-cap", formatter: "json", category: "education", value_type: "json", description: "Snapshot of an education_history row." }, + // ---- family ------------------------------------------------------------- + { predicate: "person.family_tie", label: "Family tie", icon: "heart", formatter: "json", category: "family", value_type: "json", description: "Snapshot of a family_ties row." }, + // ---- conference --------------------------------------------------------- + { predicate: "person.conference", label: "Conference", icon: "calendar", formatter: "json", category: "conference", value_type: "json", description: "Snapshot of a conference_attendance row." }, + // ---- preferences (one predicate per documented preference_key) --------- + { predicate: "person.preference.communication_channel", label: "Communication channel", icon: "message-circle", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact channel (email/text/dm/voice)." }, + { predicate: "person.preference.contact_time", label: "Best time to reach", icon: "clock", formatter: "text", category: "preference", value_type: "text", description: "Preferred contact window or day." }, + { predicate: "person.preference.meeting_format", label: "Meeting format", icon: "video", formatter: "text", category: "preference", value_type: "text", description: "In-person, video, phone, async." }, + { predicate: "person.preference.gift_dietary", label: "Dietary", icon: "salad", formatter: "text", category: "preference", value_type: "text", description: "Vegan/vegetarian/keto/halal/etc." }, + { predicate: "person.preference.gift_allergies", label: "Allergies", icon: "alert-triangle", formatter: "list", category: "preference", value_type: "json", description: "Food/material allergies to avoid for gifting." }, + { predicate: "person.preference.coffee_order", label: "Coffee order", icon: "coffee", formatter: "text", category: "preference", value_type: "text", description: "Usual coffee order." }, + { predicate: "person.preference.travel_class", label: "Travel class", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Preferred flight cabin class." }, + { predicate: "person.preference.hotel_brand", label: "Hotel brand", icon: "bed", formatter: "text", category: "preference", value_type: "text", description: "Preferred hotel chain or brand." }, + { predicate: "person.preference.airline_status", label: "Airline status", icon: "plane", formatter: "text", category: "preference", value_type: "text", description: "Frequent-flyer status / preferred carrier." }, + // ---- interests (one per category) -------------------------------------- + { predicate: "person.interest.topic", label: "Topic", icon: "tag", formatter: "badge", category: "interest", value_type: "text", description: "Topic of interest." }, + { predicate: "person.interest.sport", label: "Sport", icon: "trophy", formatter: "badge", category: "interest", value_type: "text", description: "Sport played or followed." }, + { predicate: "person.interest.team", label: "Team", icon: "shield", formatter: "badge", category: "interest", value_type: "text", description: "Favorite team." }, + { predicate: "person.interest.book", label: "Book", icon: "book", formatter: "text", category: "interest", value_type: "text", description: "Book the person recommends or read." }, + { predicate: "person.interest.author", label: "Author", icon: "feather", formatter: "text", category: "interest", value_type: "text", description: "Favorite author." }, + { predicate: "person.interest.podcast", label: "Podcast", icon: "mic", formatter: "text", category: "interest", value_type: "text", description: "Podcast the person listens to." }, + { predicate: "person.interest.music", label: "Music genre", icon: "music", formatter: "badge", category: "interest", value_type: "text", description: "Preferred music genre." }, + { predicate: "person.interest.artist", label: "Artist", icon: "music", formatter: "text", category: "interest", value_type: "text", description: "Favorite musical artist." }, + { predicate: "person.interest.film", label: "Film", icon: "film", formatter: "text", category: "interest", value_type: "text", description: "Favorite film." }, + { predicate: "person.interest.show", label: "TV show", icon: "tv", formatter: "text", category: "interest", value_type: "text", description: "Favorite TV show." }, + { predicate: "person.interest.hobby", label: "Hobby", icon: "puzzle", formatter: "badge", category: "interest", value_type: "text", description: "Hobby outside of work." }, + { predicate: "person.interest.cause", label: "Cause", icon: "heart-handshake", formatter: "badge", category: "interest", value_type: "text", description: "Cause the person supports." }, + // ---- lifestyle signals ------------------------------------------------- + { predicate: "person.lifestyle.runs", label: "Runs", icon: "footprints", formatter: "text", category: "lifestyle", value_type: "json", description: "Is a runner." }, + { predicate: "person.lifestyle.cycles", label: "Cycles", icon: "bike", formatter: "text", category: "lifestyle", value_type: "json", description: "Cycles." }, + { predicate: "person.lifestyle.surfs", label: "Surfs", icon: "waves", formatter: "text", category: "lifestyle", value_type: "json", description: "Surfs." }, + { predicate: "person.lifestyle.skis", label: "Skis", icon: "snowflake", formatter: "text", category: "lifestyle", value_type: "json", description: "Skis or snowboards." }, + { predicate: "person.lifestyle.golfs", label: "Golfs", icon: "flag", formatter: "text", category: "lifestyle", value_type: "json", description: "Plays golf." }, + { predicate: "person.lifestyle.yoga", label: "Yoga", icon: "activity", formatter: "text", category: "lifestyle", value_type: "json", description: "Practices yoga." }, + { predicate: "person.lifestyle.meditates", label: "Meditates", icon: "leaf", formatter: "text", category: "lifestyle", value_type: "json", description: "Meditates regularly." }, + { predicate: "person.lifestyle.cooks", label: "Cooks", icon: "chef-hat", formatter: "text", category: "lifestyle", value_type: "json", description: "Cooks for fun." }, + { predicate: "person.lifestyle.collects", label: "Collects", icon: "package", formatter: "text", category: "lifestyle", value_type: "json", description: "Collects something (art, wine, watches…)." }, + { predicate: "person.lifestyle.pet", label: "Pet", icon: "paw-print", formatter: "text", category: "lifestyle", value_type: "json", description: "Has a pet." }, + { predicate: "person.lifestyle.marathon", label: "Marathon", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed marathon." }, + { predicate: "person.lifestyle.ironman", label: "Ironman", icon: "medal", formatter: "text", category: "lifestyle", value_type: "json", description: "Completed Ironman triathlon." }, + // ---- travel ------------------------------------------------------------- + { predicate: "person.travel.frequent_city", label: "Frequent city", icon: "map-pin", formatter: "text", category: "travel", value_type: "text", description: "City the person frequently travels to." }, + { predicate: "person.travel.home_base", label: "Home base", icon: "home", formatter: "text", category: "travel", value_type: "text", description: "Stated home base." }, + { predicate: "person.travel.recent_trip", label: "Recent trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Recent trip place + date window." }, + { predicate: "person.travel.upcoming_trip", label: "Upcoming trip", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Announced upcoming trip." }, + { predicate: "person.travel.airport_hub", label: "Airport hub", icon: "plane", formatter: "text", category: "travel", value_type: "text", description: "Home airport (IATA code)." }, + // ---- goals -------------------------------------------------------------- + { predicate: "person.goal.short_term", label: "Short-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for <12 months." }, + { predicate: "person.goal.long_term", label: "Long-term goal", icon: "target", formatter: "text", category: "goal", value_type: "text", description: "Goal stated for 12+ months." }, + { predicate: "person.goal.hiring", label: "Hiring goal", icon: "user-plus", formatter: "text", category: "goal", value_type: "text", description: "Stated hiring need." }, + { predicate: "person.goal.fundraising", label: "Fundraising goal", icon: "dollar-sign", formatter: "text", category: "goal", value_type: "text", description: "Stated fundraising plan." }, + { predicate: "person.goal.investing_thesis", label: "Investing thesis", icon: "lightbulb", formatter: "text", category: "goal", value_type: "text", description: "Stated investing thesis." }, + { predicate: "person.goal.expansion_market", label: "Expansion market", icon: "map", formatter: "text", category: "goal", value_type: "text", description: "Stated market expansion." }, + // ---- conversation hooks ------------------------------------------------ + { predicate: "person.hook.recent_news", label: "Recent news", icon: "newspaper", formatter: "text", category: "hook", value_type: "text", description: "Recent news about the person." }, + { predicate: "person.hook.shared_connection", label: "Shared connection", icon: "users", formatter: "text", category: "hook", value_type: "text", description: "Mutual contact you can mention." }, + { predicate: "person.hook.shared_school", label: "Shared school", icon: "graduation-cap", formatter: "text", category: "hook", value_type: "text", description: "Shared alma mater." }, + { predicate: "person.hook.shared_employer", label: "Shared employer", icon: "briefcase", formatter: "text", category: "hook", value_type: "text", description: "Shared former or current employer." }, + { predicate: "person.hook.shared_interest", label: "Shared interest", icon: "tag", formatter: "text", category: "hook", value_type: "text", description: "Shared interest or hobby." }, + { predicate: "person.hook.recent_post", label: "Recent post", icon: "message-square", formatter: "text", category: "hook", value_type: "text", description: "Recent public post by the person." }, + { predicate: "person.hook.life_event", label: "Life event", icon: "sparkles", formatter: "text", category: "hook", value_type: "text", description: "Life event (job change, baby, move)." }, + { predicate: "person.hook.opinion_quoted", label: "Opinion quoted", icon: "quote", formatter: "text", category: "hook", value_type: "text", description: "Public statement on a topic." }, + // ---- appreciation ------------------------------------------------------ + { predicate: "person.appreciation.compliment_topic", label: "Compliment topic", icon: "thumbs-up", formatter: "text", category: "appreciation", value_type: "text", description: "Topic the person likes to be complimented on." }, + { predicate: "person.appreciation.gift_idea", label: "Gift idea", icon: "gift", formatter: "text", category: "appreciation", value_type: "text", description: "Concrete gift idea." }, + { predicate: "person.appreciation.charity_supported", label: "Charity supported", icon: "heart", formatter: "text", category: "appreciation", value_type: "text", description: "Charity the person supports." }, + { predicate: "person.appreciation.cause_advocated", label: "Cause advocated", icon: "megaphone", formatter: "text", category: "appreciation", value_type: "text", description: "Cause the person publicly advocates." }, + { predicate: "person.appreciation.recognition_received", label: "Recognition received", icon: "award", formatter: "text", category: "appreciation", value_type: "text", description: "Public award/recognition received." }, + // ---- legacy / extractor predicates (so the UI never renders a raw key) - + { predicate: "name", label: "Name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Display name (extractor field)." }, + { predicate: "title", label: "Title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Job title (extractor field)." }, + { predicate: "role", label: "Role", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Functional role." }, + { predicate: "employer", label: "Employer", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Current employer name." }, + { predicate: "company", label: "Company", icon: "building", formatter: "text", category: "career", value_type: "text", description: "Alias for employer in some extractors." }, + { predicate: "headline", label: "Headline", icon: "type", formatter: "text", category: "identity", value_type: "text", description: "LinkedIn-style headline." }, + { predicate: "summary", label: "Summary", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Bio / summary paragraph." }, + { predicate: "bio", label: "Bio", icon: "file-text", formatter: "text", category: "identity", value_type: "text", description: "Profile bio." }, + { predicate: "description", label: "Description", icon: "file-text", formatter: "text", category: "firm", value_type: "text", description: "Org/company description." }, + { predicate: "location", label: "Location", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "Stated location string." }, + { predicate: "city", label: "City", icon: "map-pin", formatter: "text", category: "identity", value_type: "text", description: "City (extractor field)." }, + { predicate: "region", label: "Region", icon: "map", formatter: "text", category: "identity", value_type: "text", description: "State/region." }, + { predicate: "country", label: "Country", icon: "globe", formatter: "text", category: "identity", value_type: "text", description: "Country (name string)." }, + { predicate: "country_iso2", label: "Country (ISO)", icon: "flag", formatter: "flag", category: "identity", value_type: "country_iso2", description: "ISO 3166-1 alpha-2 country." }, + { predicate: "timezone", label: "Timezone", icon: "clock", formatter: "text", category: "identity", value_type: "text", description: "Timezone (extractor field)." }, + { predicate: "email", label: "Email", icon: "mail", formatter: "link", category: "contact", value_type: "email", description: "Email address." }, + { predicate: "phone", label: "Phone", icon: "phone", formatter: "text", category: "contact", value_type: "text", description: "Phone number (E.164 preferred)." }, + { predicate: "linkedin_url", label: "LinkedIn", icon: "linkedin", formatter: "link", category: "contact", value_type: "url", description: "LinkedIn profile URL." }, + { predicate: "twitter_url", label: "Twitter", icon: "twitter", formatter: "link", category: "contact", value_type: "url", description: "Twitter/X profile URL." }, + { predicate: "twitter_handle", label: "Twitter handle", icon: "twitter", formatter: "text", category: "contact", value_type: "text", description: "Twitter/X handle." }, + { predicate: "github_url", label: "GitHub", icon: "github", formatter: "link", category: "contact", value_type: "url", description: "GitHub profile URL." }, + { predicate: "github_handle", label: "GitHub handle", icon: "github", formatter: "text", category: "contact", value_type: "text", description: "GitHub handle." }, + { predicate: "website", label: "Website", icon: "link", formatter: "link", category: "contact", value_type: "url", description: "Personal/firm website URL." }, + { predicate: "primary_domain", label: "Primary domain", icon: "globe", formatter: "text", category: "firm", value_type: "text", description: "Canonical apex domain." }, + { predicate: "sector", label: "Sector", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Sector / vertical tag." }, + { predicate: "stage", label: "Stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Investment stage focus." }, + { predicate: "check_size_min_usd", label: "Min check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Minimum typical check size." }, + { predicate: "check_size_max_usd", label: "Max check (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Maximum typical check size." }, + { predicate: "fund_size_usd", label: "Fund size (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent fund size." }, + { predicate: "founded_year", label: "Founded year", icon: "calendar", formatter: "year", category: "firm", value_type: "year", description: "Year founded." }, + { predicate: "founded_at", label: "Founded", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Founding date (full)." }, + { predicate: "funding_stage", label: "Funding stage", icon: "trending-up", formatter: "badge", category: "firm", value_type: "text", description: "Latest funding stage." }, + { predicate: "total_funding_usd", label: "Total funding (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Total funding raised." }, + { predicate: "last_round_usd", label: "Last round (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Most recent round amount." }, + { predicate: "hq_city", label: "HQ city", icon: "map-pin", formatter: "text", category: "firm", value_type: "text", description: "HQ city." }, + { predicate: "hq_country_iso2", label: "HQ country (ISO)", icon: "flag", formatter: "flag", category: "firm", value_type: "country_iso2", description: "HQ country ISO code." }, + { predicate: "employees", label: "Employees", icon: "users", formatter: "text", category: "firm", value_type: "number", description: "Employee headcount or band." }, + { predicate: "industry", label: "Industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry classification." }, + { predicate: "display_name", label: "Display name", icon: "user", formatter: "text", category: "identity", value_type: "text", description: "Canonical display name." }, + // ---- Task #1: SEC EDGAR deep-adapter predicates ----------------------- + // Mirror in migration 349_sec_edgar.sql — two-file change enforced by + // test/profile.test.mjs. + { predicate: "sec.cik", label: "SEC CIK", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC Central Index Key (10-digit, zero-padded)." }, + { predicate: "sec.crd", label: "SEC CRD", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Investment Adviser CRD# from Form ADV." }, + { predicate: "sec.cusip", label: "CUSIP", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "Committee on Uniform Securities Identification Procedures code." }, + { predicate: "sec.ticker", label: "Ticker", icon: "trending-up", formatter: "badge", category: "identity", value_type: "text", description: "Public ticker symbol." }, + { predicate: "sec.sec_file_number", label: "SEC file number", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC file number (e.g. 801-12345 for advisers)." }, + { predicate: "sec.fund_id_807", label: "SEC fund ID", icon: "hash", formatter: "text", category: "identity", value_type: "text", description: "SEC fund identifier (807-XXXXXXXX)." }, + { predicate: "aum_usd", label: "AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Assets under management in USD." }, + { predicate: "sec.form_adv.filed_at", label: "Last Form ADV filed", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Most recent Form ADV acceptance date." }, + { predicate: "sec.form_adv.fund", label: "Form ADV fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Fund disclosed on Schedule D §7.B.(1)." }, + { predicate: "sec.form_d.round", label: "Form D round", icon: "dollar-sign", formatter: "json", category: "firm", value_type: "json", description: "Private placement disclosed on Form D." }, + { predicate: "sec.form_d.issuer_industry", label: "Form D industry", icon: "layers", formatter: "badge", category: "firm", value_type: "text", description: "Industry group declared on Form D." }, + { predicate: "sec.13f.holding", label: "13F holding", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Equity position disclosed on Form 13F-HR." }, + { predicate: "sec.13f.filer_aum_usd", label: "13F filer AUM (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Aggregate USD value of 13F holdings (proxy AUM)." }, + { predicate: "sec.13d.beneficial_owner", label: "13D beneficial owner", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Schedule 13D 5%+ beneficial-ownership disclosure." }, + { predicate: "sec.form4.insider_trade", label: "Insider trade", icon: "arrow-up-down", formatter: "json", category: "firm", value_type: "json", description: "Form 4 §16 insider transaction." }, + { predicate: "sec.form4.officer_title", label: "Officer title", icon: "briefcase", formatter: "text", category: "career", value_type: "text", description: "Officer title declared on Form 4 (when reporter is officer)." }, + { predicate: "sec.s1.ipo_intent", label: "S-1 IPO intent", icon: "rocket", formatter: "text", category: "firm", value_type: "text", description: "Company filed Form S-1 (IPO registration)." }, + { predicate: "sec.s1.underwriter", label: "IPO underwriter", icon: "briefcase", formatter: "text", category: "firm", value_type: "text", description: "Underwriter listed on Form S-1." }, + { predicate: "sec.8k.material_event", label: "8-K material event", icon: "alert-circle", formatter: "json", category: "firm", value_type: "json", description: "Form 8-K current report item." }, + { predicate: "sec.10k.revenue_usd", label: "10-K revenue (USD)", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Annual revenue from Form 10-K." }, + { predicate: "sec.10k.net_income_usd", label: "10-K net income", icon: "dollar-sign", formatter: "usd", category: "firm", value_type: "currency_usd", description: "Net income from Form 10-K." }, + { predicate: "sec.10k.fiscal_year_end", label: "10-K fiscal year-end", icon: "calendar", formatter: "date", category: "firm", value_type: "date", description: "Fiscal year-end from Form 10-K." }, + { predicate: "sec.10k.executive", label: "10-K executive", icon: "users", formatter: "json", category: "firm", value_type: "json", description: "Named executive officer compensation from Form 10-K." }, + { predicate: "sec.pf.fund", label: "Form PF fund", icon: "briefcase", formatter: "json", category: "firm", value_type: "json", description: "Private fund disclosure from Form PF (large private fund adviser)." }, + { predicate: "sec.gp_disclosed", label: "GP disclosed (SEC)", icon: "user-check", formatter: "text", category: "firm", value_type: "text", description: "GP / control-person disclosed on a SEC filing." }, +]; +export const PREDICATE_MAP = Object.freeze(Object.fromEntries(PREDICATE_REGISTRY.map((p) => [p.predicate, p]))); +export function getPredicateMeta(predicate) { + return PREDICATE_MAP[predicate] ?? null; +} +// Helpers in `profile.ts` MUST only emit predicates listed here. +// The smoke test asserts EMITTED_PREDICATES ⊆ PREDICATE_REGISTRY keys. +export const EMITTED_PREDICATES = Object.freeze([ + // static (one per helper) + "person.identity", + "person.career", + "person.board_seat", + "person.education", + "person.family_tie", + "person.conference", + // dynamic: person.preference.{key} + "person.preference.communication_channel", + "person.preference.contact_time", + "person.preference.meeting_format", + "person.preference.gift_dietary", + "person.preference.gift_allergies", + "person.preference.coffee_order", + "person.preference.travel_class", + "person.preference.hotel_brand", + "person.preference.airline_status", + // dynamic: person.interest.{category} + "person.interest.topic", "person.interest.sport", "person.interest.team", + "person.interest.book", "person.interest.author", "person.interest.podcast", + "person.interest.music", "person.interest.artist", "person.interest.film", + "person.interest.show", "person.interest.hobby", "person.interest.cause", + // dynamic: person.lifestyle.{key} + "person.lifestyle.runs", "person.lifestyle.cycles", "person.lifestyle.surfs", + "person.lifestyle.skis", "person.lifestyle.golfs", "person.lifestyle.yoga", + "person.lifestyle.meditates", "person.lifestyle.cooks", "person.lifestyle.collects", + "person.lifestyle.pet", "person.lifestyle.marathon", "person.lifestyle.ironman", + // dynamic: person.travel.{kind} + "person.travel.frequent_city", "person.travel.home_base", + "person.travel.recent_trip", "person.travel.upcoming_trip", "person.travel.airport_hub", + // dynamic: person.goal.{kind} + "person.goal.short_term", "person.goal.long_term", "person.goal.hiring", + "person.goal.fundraising", "person.goal.investing_thesis", "person.goal.expansion_market", + // dynamic: person.hook.{kind} + "person.hook.recent_news", "person.hook.shared_connection", "person.hook.shared_school", + "person.hook.shared_employer", "person.hook.shared_interest", "person.hook.recent_post", + "person.hook.life_event", "person.hook.opinion_quoted", + // dynamic: person.appreciation.{kind} + "person.appreciation.compliment_topic", "person.appreciation.gift_idea", + "person.appreciation.charity_supported", "person.appreciation.cause_advocated", + "person.appreciation.recognition_received", +]); diff --git a/apps/worker/test-dist-q/entities/profile-shapes.js b/apps/worker/test-dist-q/entities/profile-shapes.js new file mode 100644 index 00000000..ab1d4f7f --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile-shapes.js @@ -0,0 +1,4 @@ +// Task #4: Typed shapes for the JSON columns on rich-profile tables. +// Every helper in `profile.ts` serializes through these shapes; nothing +// passes raw `unknown` through `JSON.stringify`. +export {}; diff --git a/apps/worker/test-dist-q/entities/profile.js b/apps/worker/test-dist-q/entities/profile.js new file mode 100644 index 00000000..7de9b64d --- /dev/null +++ b/apps/worker/test-dist-q/entities/profile.js @@ -0,0 +1,681 @@ +// Task #4: EntityService write helpers for the rich person profile. +// +// Every helper: +// 1. Validates input against the typed shape (profile-shapes.ts). +// 2. Acquires the per-entity DO lock (EntityLock /acquire) so concurrent +// scrapers / OSINT / agent calls don't race on the same entity_id. +// 3. Upserts the structured row using a stable natural key +// (ON CONFLICT(...) DO UPDATE). +// 4. Mirrors a canonical row into `facts` via insertFact — the fact's +// hash dedupe key is sha256(entity|predicate|value|source_url) so the +// second call with identical content updates `observed_at` rather +// than creating a duplicate row. +// +// Public-signal constraint: every helper except `setPersonIdentity` (which +// allows operator-asserted rows with isOperatorAsserted=true) refuses to +// write without a source_url. +import { sha256 } from "./normalize"; +import { enqueueSummaryRebuild } from "./summaryQueue"; +import { EMITTED_PREDICATES, PREDICATE_MAP, } from "./profile-predicates"; +// ---- Internal: per-entity lock (token mutex via EntityLock DO) ----------- +// +// Mirrors the OSINT resolver's acquire/release pattern so all rich-profile +// writes for a given entity_id serialize against OSINT, dual-write, and +// any other caller that already uses the same DO id-namespace. +// +// Strict serialization (task contract): when the ENTITY_LOCK binding is +// present we MUST hold the lock for the duration of the write. If the +// DO is unreachable or the lock is currently held by another writer we +// retry with exponential backoff and then fail loudly rather than +// silently racing. The only bypass is when the binding itself is absent +// — that's how the in-memory unit tests run, and it's explicit. +export class ProfileLockError extends Error { + constructor(entityId, reason) { + super(`profile lock for ${entityId}: ${reason}`); + this.name = "ProfileLockError"; + } +} +async function withProfileLock(env, entityId, fn) { + if (!env.ENTITY_LOCK) + return await fn(); + const stub = env.ENTITY_LOCK.get(env.ENTITY_LOCK.idFromName(entityId)); + const token = crypto.randomUUID(); + let acquired = false; + let lastErr = "unknown"; + // 5 attempts, 50 / 100 / 200 / 400 / 800ms backoff = ≤1.55s total. + // Beyond that we surface the error so the caller can decide (retry the + // job, surface to the operator, etc.) rather than corrupt with a + // racy write. + for (let attempt = 0; attempt < 5 && !acquired; attempt++) { + if (attempt > 0) { + await new Promise((r) => setTimeout(r, 50 * 2 ** (attempt - 1))); + } + try { + const res = await stub.fetch("https://lock/acquire", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ token, ttlMs: 60_000 }), + }); + if (res.ok) { + acquired = true; + break; + } + lastErr = `acquire returned ${res.status}`; + } + catch (e) { + lastErr = e.message || "fetch failed"; + } + } + if (!acquired) { + throw new ProfileLockError(entityId, lastErr); + } + try { + return await fn(); + } + finally { + try { + await stub.fetch("https://lock/release", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ token }), + }); + } + catch { /* lock will TTL out after 60s */ } + } +} +// ---- Internal: mirror a structured row into `facts` --------------------- +// +// Centralized projection so every helper produces consistent fact rows. +// +// Dedupe contract (task spec): natural key is (entity_id, predicate, +// source_url) — independent of value. Re-observing the same predicate +// from the same source_url MUST upsert the existing fact (new value, +// refreshed observed_at) rather than create a second row. The hash we +// compute here covers ONLY those three fields, and UNIQUE(hash) on +// `facts` (migration 201) turns a collision into our UPDATE path. +// +// We bypass `insertFact` for the mirror path because its hash includes +// the value (which is correct for raw scraper writes but wrong here — +// it would let value drift create duplicate (entity, predicate, source) +// rows). The summary-rebuild enqueue is still triggered. +// +// Asserts the predicate exists in the registry — catches typos at +// runtime and is defense-in-depth backup to the smoke-test enforcement +// of EMITTED_PREDICATES ⊆ registry. +async function mirrorFact(env, args) { + if (!PREDICATE_MAP[args.predicate]) { + throw new Error(`profile.mirrorFact: predicate "${args.predicate}" is not in PREDICATE_REGISTRY`); + } + const hash = await sha256(`${args.entityId}|${args.predicate}|${args.sourceUrl}`); + const valueText = args.valueText ?? null; + const valueNumber = args.valueNumber ?? null; + const valueJsonStr = args.valueJson != null ? JSON.stringify(args.valueJson) : null; + const confidence = args.confidence ?? 1.0; + const now = args.observedAt ?? new Date().toISOString(); + try { + await env.DB.prepare(`INSERT INTO facts ( + id, entity_id, predicate, value_text, value_number, value_json, + value_entity_id, source_kind, source, evidence_url, confidence, + observed_at, valid_from, valid_to, is_current, hash + ) VALUES (?, ?, ?, ?, ?, ?, NULL, 'enrichment', ?, ?, ?, ?, NULL, NULL, 1, ?)`).bind(crypto.randomUUID(), args.entityId, args.predicate, valueText, valueNumber, valueJsonStr, args.sourceUrl, args.sourceUrl, confidence, now, hash).run(); + } + catch (e) { + const msg = e.message || ""; + if (/UNIQUE/i.test(msg)) { + await env.DB.prepare(`UPDATE facts + SET value_text = ?, value_number = ?, value_json = ?, + confidence = MAX(confidence, ?), + observed_at = ?, is_current = 1 + WHERE hash = ?`).bind(valueText, valueNumber, valueJsonStr, confidence, now, hash).run(); + } + else { + throw e; + } + } + try { + await enqueueSummaryRebuild(env, args.entityId); + } + catch { /* best-effort */ } +} +function requireSourceUrl(helper, url) { + if (!url || typeof url !== "string" || url.trim().length === 0) { + throw new Error(`profile.${helper}: source_url is required (public-signal-only constraint)`); + } + return url; +} +function requireNonEmpty(helper, field, v) { + if (!v || typeof v !== "string" || v.trim().length === 0) { + throw new Error(`profile.${helper}: ${field} is required`); + } + return v.trim(); +} +function nowIso() { return new Date().toISOString(); } +// ========================================================================= +// 1. setPersonIdentity — upsert on entity_id. +// ========================================================================= +export async function setPersonIdentity(env, input) { + requireNonEmpty("setPersonIdentity", "entityId", input.entityId); + const isOperator = input.isOperatorAsserted === true; + if (!isOperator) + requireSourceUrl("setPersonIdentity", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_identity ( + entity_id, full_name, preferred_name, pronouns_json, birth_year, + nationality, languages_json, timezone, location_city, location_country, + headshot_url, source_url, is_operator_asserted, confidence, + observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id) DO UPDATE SET + full_name = COALESCE(excluded.full_name, person_identity.full_name), + preferred_name = COALESCE(excluded.preferred_name, person_identity.preferred_name), + pronouns_json = COALESCE(excluded.pronouns_json, person_identity.pronouns_json), + birth_year = COALESCE(excluded.birth_year, person_identity.birth_year), + nationality = COALESCE(excluded.nationality, person_identity.nationality), + languages_json = COALESCE(excluded.languages_json, person_identity.languages_json), + timezone = COALESCE(excluded.timezone, person_identity.timezone), + location_city = COALESCE(excluded.location_city, person_identity.location_city), + location_country = COALESCE(excluded.location_country, person_identity.location_country), + headshot_url = COALESCE(excluded.headshot_url, person_identity.headshot_url), + source_url = COALESCE(excluded.source_url, person_identity.source_url), + is_operator_asserted = MAX(excluded.is_operator_asserted, person_identity.is_operator_asserted), + confidence = MAX(excluded.confidence, person_identity.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(input.entityId, input.fullName ?? null, input.preferredName ?? null, input.pronouns ? JSON.stringify(input.pronouns) : null, input.birthYear ?? null, input.nationality ?? null, input.languages ? JSON.stringify(input.languages) : null, input.timezone ?? null, input.locationCity ?? null, input.locationCountry ?? null, input.headshotUrl ?? null, input.sourceUrl ?? null, isOperator ? 1 : 0, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.identity", + sourceUrl: input.sourceUrl ?? "operator://asserted", + valueJson: { + full_name: input.fullName ?? null, + preferred_name: input.preferredName ?? null, + pronouns: input.pronouns ?? null, + birth_year: input.birthYear ?? null, + nationality: input.nationality ?? null, + languages: input.languages ?? null, + timezone: input.timezone ?? null, + location_city: input.locationCity ?? null, + location_country: input.locationCountry ?? null, + headshot_url: input.headshotUrl ?? null, + is_operator_asserted: isOperator, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 2. addCareerEntry — natural key (entity_id, organization_*, started_at). +// ========================================================================= +export async function addCareerEntry(env, input) { + requireNonEmpty("addCareerEntry", "entityId", input.entityId); + requireNonEmpty("addCareerEntry", "organizationName", input.organizationName); + const sourceUrl = requireSourceUrl("addCareerEntry", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO career_history ( + id, entity_id, organization_entity_id, organization_name, role_title, + seniority, department, started_at, ended_at, is_current, summary, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, COALESCE(organization_entity_id,''), COALESCE(started_at,'')) DO UPDATE SET + organization_name = COALESCE(excluded.organization_name, career_history.organization_name), + role_title = COALESCE(excluded.role_title, career_history.role_title), + seniority = COALESCE(excluded.seniority, career_history.seniority), + department = COALESCE(excluded.department, career_history.department), + ended_at = COALESCE(excluded.ended_at, career_history.ended_at), + is_current = excluded.is_current, + summary = COALESCE(excluded.summary, career_history.summary), + confidence = MAX(excluded.confidence, career_history.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.roleTitle ?? null, input.seniority ?? null, input.department ?? null, input.startedAt ?? null, input.endedAt ?? null, input.isCurrent ? 1 : 0, input.summary ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.career", + sourceUrl, + valueJson: { + organization_entity_id: input.organizationEntityId ?? null, + organization_name: input.organizationName, + role_title: input.roleTitle ?? null, + seniority: input.seniority ?? null, + department: input.department ?? null, + started_at: input.startedAt ?? null, + ended_at: input.endedAt ?? null, + is_current: input.isCurrent === true, + }, + confidence: input.confidence, + observedAt: now, + }); + }); + // Task #8: career changes materially affect persona match scores + // (title, seniority, function, employer industry/size/stage). Fire + // the debounced re-match trigger. + try { + const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); + void triggerEntityMatchRefresh(env, input.entityId).catch(() => undefined); + } + catch { /* trigger is best-effort */ } +} +// ========================================================================= +// 3. addBoardSeat — natural key (entity_id, organization_name, started_at). +// ========================================================================= +export async function addBoardSeat(env, input) { + requireNonEmpty("addBoardSeat", "entityId", input.entityId); + requireNonEmpty("addBoardSeat", "organizationName", input.organizationName); + const sourceUrl = requireSourceUrl("addBoardSeat", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO board_seats ( + id, entity_id, organization_entity_id, organization_name, role, + is_independent, committee, started_at, ended_at, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, organization_name, COALESCE(started_at,'')) DO UPDATE SET + organization_entity_id = COALESCE(excluded.organization_entity_id, board_seats.organization_entity_id), + role = COALESCE(excluded.role, board_seats.role), + is_independent = excluded.is_independent, + committee = COALESCE(excluded.committee, board_seats.committee), + ended_at = COALESCE(excluded.ended_at, board_seats.ended_at), + confidence = MAX(excluded.confidence, board_seats.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.organizationEntityId ?? null, input.organizationName, input.role ?? null, input.isIndependent ? 1 : 0, input.committee ?? null, input.startedAt ?? null, input.endedAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.board_seat", + sourceUrl, + valueJson: { + organization_entity_id: input.organizationEntityId ?? null, + organization_name: input.organizationName, + role: input.role ?? null, + is_independent: input.isIndependent === true, + committee: input.committee ?? null, + started_at: input.startedAt ?? null, + ended_at: input.endedAt ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 4. addEducation — natural key (entity_id, institution, degree, ended_year). +// ========================================================================= +export async function addEducation(env, input) { + requireNonEmpty("addEducation", "entityId", input.entityId); + requireNonEmpty("addEducation", "institution", input.institution); + const sourceUrl = requireSourceUrl("addEducation", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO education_history ( + id, entity_id, institution, degree, field, started_year, ended_year, + honors, source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, institution, COALESCE(degree,''), COALESCE(ended_year,0)) DO UPDATE SET + field = COALESCE(excluded.field, education_history.field), + started_year = COALESCE(excluded.started_year, education_history.started_year), + honors = COALESCE(excluded.honors, education_history.honors), + confidence = MAX(excluded.confidence, education_history.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.institution, input.degree ?? null, input.field ?? null, input.startedYear ?? null, input.endedYear ?? null, input.honors ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.education", + sourceUrl, + valueJson: { + institution: input.institution, + degree: input.degree ?? null, + field: input.field ?? null, + started_year: input.startedYear ?? null, + ended_year: input.endedYear ?? null, + honors: input.honors ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 5. addFamilyTie — natural key (entity_id, relation_type, related_name). +// +// Privacy gate (task contract, "No PII leakage"): +// * `isPublic` is required — no defaulting (the helper throws if it's +// undefined / not a boolean), so a caller can never implicitly create +// a private row by omission. +// * `isPublic === false` additionally requires `isOperatorAsserted === +// true`. This is the explicit operator-intent marker the task calls +// out: only the human operator (via an authenticated route handler +// that sets this flag) can stash a private relationship. Background +// enrichment, agents, and scrapers will never have it set and will +// be rejected. +// * Private rows are NEVER mirrored into `facts` — `facts` is the +// public/agent retrieval surface, and routing PII through it would +// undermine row-level filtering downstream. The structured +// `family_ties` row is the only store for private ties, and the +// route layer is responsible for gating reads by operator identity. +// ========================================================================= +export async function addFamilyTie(env, input) { + requireNonEmpty("addFamilyTie", "entityId", input.entityId); + requireNonEmpty("addFamilyTie", "relationType", input.relationType); + requireNonEmpty("addFamilyTie", "relatedName", input.relatedName); + if (typeof input.isPublic !== "boolean") { + throw new Error("profile.addFamilyTie: isPublic is required and must be an explicit boolean"); + } + if (input.isPublic === false && input.isOperatorAsserted !== true) { + throw new Error("profile.addFamilyTie: private family ties (isPublic=false) require " + + "isOperatorAsserted=true; background enrichers and agents cannot store private relationships"); + } + const sourceUrl = requireSourceUrl("addFamilyTie", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO family_ties ( + id, entity_id, relation_type, related_name, related_entity_id, notes, + is_public, source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, relation_type, related_name) DO UPDATE SET + related_entity_id = COALESCE(excluded.related_entity_id, family_ties.related_entity_id), + notes = COALESCE(excluded.notes, family_ties.notes), + is_public = excluded.is_public, + confidence = MAX(excluded.confidence, family_ties.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.relationType, input.relatedName, input.relatedEntityId ?? null, input.notes ?? null, input.isPublic ? 1 : 0, sourceUrl, input.confidence ?? 1.0, now, now).run(); + // PII firewall: only public ties are projected to `facts`. The + // structured row above is still written for the operator-only UI. + if (input.isPublic === true) { + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.family_tie", + sourceUrl, + valueJson: { + relation_type: input.relationType, + related_name: input.relatedName, + related_entity_id: input.relatedEntityId ?? null, + notes: input.notes ?? null, + is_public: true, + }, + confidence: input.confidence, + observedAt: now, + }); + } + }); +} +// ========================================================================= +// 6. addPreference — upsert on (entity_id, preference_key). +// Mirrors to person.preference.{preferenceKey}; that dynamic predicate +// MUST exist in the registry (validated in mirrorFact + smoke test). +// ========================================================================= +export async function addPreference(env, input) { + requireNonEmpty("addPreference", "entityId", input.entityId); + requireNonEmpty("addPreference", "preferenceKey", input.preferenceKey); + const sourceUrl = requireSourceUrl("addPreference", input.sourceUrl); + const predicate = `person.preference.${input.preferenceKey}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_preferences ( + id, entity_id, preference_key, value_text, value_json, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, preference_key) DO UPDATE SET + value_text = COALESCE(excluded.value_text, person_preferences.value_text), + value_json = COALESCE(excluded.value_json, person_preferences.value_json), + source_url = excluded.source_url, + confidence = MAX(excluded.confidence, person_preferences.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.preferenceKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.valueText ?? null, + valueJson: input.valueJson ?? null, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 7. addInterest — natural key (entity_id, interest_category, interest_value). +// ========================================================================= +export async function addInterest(env, input) { + requireNonEmpty("addInterest", "entityId", input.entityId); + requireNonEmpty("addInterest", "interestCategory", input.interestCategory); + requireNonEmpty("addInterest", "interestValue", input.interestValue); + const sourceUrl = requireSourceUrl("addInterest", input.sourceUrl); + const predicate = `person.interest.${input.interestCategory}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_interests ( + id, entity_id, interest_category, interest_value, weight, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, interest_category, interest_value) DO UPDATE SET + weight = MAX(excluded.weight, person_interests.weight), + confidence = MAX(excluded.confidence, person_interests.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.interestCategory, input.interestValue, input.weight ?? 1.0, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.interestValue, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 8. addLifestyleSignal — natural key (entity_id, signal_key, observed_at). +// ========================================================================= +export async function addLifestyleSignal(env, input) { + requireNonEmpty("addLifestyleSignal", "entityId", input.entityId); + requireNonEmpty("addLifestyleSignal", "signalKey", input.signalKey); + const sourceUrl = requireSourceUrl("addLifestyleSignal", input.sourceUrl); + const predicate = `person.lifestyle.${input.signalKey}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO lifestyle_signals ( + id, entity_id, signal_key, value_text, value_json, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, signal_key) DO UPDATE SET + value_text = COALESCE(excluded.value_text, lifestyle_signals.value_text), + value_json = COALESCE(excluded.value_json, lifestyle_signals.value_json), + source_url = excluded.source_url, + confidence = MAX(excluded.confidence, lifestyle_signals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKey, input.valueText ?? null, input.valueJson ? JSON.stringify(input.valueJson) : null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.valueText ?? null, + valueJson: input.valueJson ?? null, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 9. addTravelPattern — natural key (entity_id, pattern_kind, place, starts_at). +// ========================================================================= +export async function addTravelPattern(env, input) { + requireNonEmpty("addTravelPattern", "entityId", input.entityId); + requireNonEmpty("addTravelPattern", "patternKind", input.patternKind); + requireNonEmpty("addTravelPattern", "place", input.place); + const sourceUrl = requireSourceUrl("addTravelPattern", input.sourceUrl); + const predicate = `person.travel.${input.patternKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO travel_patterns ( + id, entity_id, pattern_kind, place, country_iso2, starts_at, ends_at, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, pattern_kind, place, COALESCE(starts_at,'')) DO UPDATE SET + country_iso2 = COALESCE(excluded.country_iso2, travel_patterns.country_iso2), + ends_at = COALESCE(excluded.ends_at, travel_patterns.ends_at), + confidence = MAX(excluded.confidence, travel_patterns.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.patternKind, input.place, input.countryIso2 ?? null, input.startsAt ?? null, input.endsAt ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.place, + valueJson: { + country_iso2: input.countryIso2 ?? null, + starts_at: input.startsAt ?? null, + ends_at: input.endsAt ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 10. addConferenceAttendance — UNIQUE(entity_id, conference_name, year). +// ========================================================================= +export async function addConferenceAttendance(env, input) { + requireNonEmpty("addConferenceAttendance", "entityId", input.entityId); + requireNonEmpty("addConferenceAttendance", "conferenceName", input.conferenceName); + if (!Number.isInteger(input.year) || input.year < 1900 || input.year > 2100) { + throw new Error("profile.addConferenceAttendance: year must be a 4-digit integer"); + } + const sourceUrl = requireSourceUrl("addConferenceAttendance", input.sourceUrl); + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO conference_attendance ( + id, entity_id, conference_name, year, role, session_topic, city, country_iso2, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, conference_name, year) DO UPDATE SET + role = COALESCE(excluded.role, conference_attendance.role), + session_topic = COALESCE(excluded.session_topic, conference_attendance.session_topic), + city = COALESCE(excluded.city, conference_attendance.city), + country_iso2 = COALESCE(excluded.country_iso2, conference_attendance.country_iso2), + confidence = MAX(excluded.confidence, conference_attendance.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.conferenceName, input.year, input.role ?? null, input.sessionTopic ?? null, input.city ?? null, input.countryIso2 ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate: "person.conference", + sourceUrl, + valueJson: { + conference_name: input.conferenceName, + year: input.year, + role: input.role ?? null, + session_topic: input.sessionTopic ?? null, + city: input.city ?? null, + country_iso2: input.countryIso2 ?? null, + }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 11. addGoal — natural key (entity_id, goal_kind, goal_text). +// ========================================================================= +export async function addGoal(env, input) { + requireNonEmpty("addGoal", "entityId", input.entityId); + requireNonEmpty("addGoal", "goalKind", input.goalKind); + requireNonEmpty("addGoal", "goalText", input.goalText); + const sourceUrl = requireSourceUrl("addGoal", input.sourceUrl); + const predicate = `person.goal.${input.goalKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO person_goals ( + id, entity_id, goal_kind, goal_text, target_date, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, goal_kind, goal_text) DO UPDATE SET + target_date = COALESCE(excluded.target_date, person_goals.target_date), + confidence = MAX(excluded.confidence, person_goals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.goalKind, input.goalText, input.targetDate ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.goalText, + valueJson: { target_date: input.targetDate ?? null }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 12. addConversationHook — natural key (entity_id, hook_kind, hook_text). +// ========================================================================= +export async function addConversationHook(env, input) { + requireNonEmpty("addConversationHook", "entityId", input.entityId); + requireNonEmpty("addConversationHook", "hookKind", input.hookKind); + requireNonEmpty("addConversationHook", "hookText", input.hookText); + const sourceUrl = requireSourceUrl("addConversationHook", input.sourceUrl); + const predicate = `person.hook.${input.hookKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO conversation_hooks ( + id, entity_id, hook_kind, hook_text, related_entity_id, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, hook_kind, hook_text) DO UPDATE SET + related_entity_id = COALESCE(excluded.related_entity_id, conversation_hooks.related_entity_id), + confidence = MAX(excluded.confidence, conversation_hooks.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.hookKind, input.hookText, input.relatedEntityId ?? null, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.hookText, + valueJson: { related_entity_id: input.relatedEntityId ?? null }, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// ========================================================================= +// 13. addAppreciationSignal — natural key (entity_id, signal_kind, signal_text). +// ========================================================================= +export async function addAppreciationSignal(env, input) { + requireNonEmpty("addAppreciationSignal", "entityId", input.entityId); + requireNonEmpty("addAppreciationSignal", "signalKind", input.signalKind); + requireNonEmpty("addAppreciationSignal", "signalText", input.signalText); + const sourceUrl = requireSourceUrl("addAppreciationSignal", input.sourceUrl); + const predicate = `person.appreciation.${input.signalKind}`; + const now = input.observedAt ?? nowIso(); + await withProfileLock(env, input.entityId, async () => { + await env.DB.prepare(`INSERT INTO appreciation_signals ( + id, entity_id, signal_kind, signal_text, + source_url, confidence, observed_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id, signal_kind, signal_text) DO UPDATE SET + confidence = MAX(excluded.confidence, appreciation_signals.confidence), + observed_at = excluded.observed_at, + updated_at = excluded.updated_at`).bind(crypto.randomUUID(), input.entityId, input.signalKind, input.signalText, sourceUrl, input.confidence ?? 1.0, now, now).run(); + await mirrorFact(env, { + entityId: input.entityId, + predicate, + sourceUrl, + valueText: input.signalText, + confidence: input.confidence, + observedAt: now, + }); + }); +} +// EntityService facade — single import surface for callers. +export const EntityService = { + setPersonIdentity, + addCareerEntry, + addBoardSeat, + addEducation, + addFamilyTie, + addPreference, + addInterest, + addLifestyleSignal, + addTravelPattern, + addConferenceAttendance, + addGoal, + addConversationHook, + addAppreciationSignal, +}; +export { EMITTED_PREDICATES }; diff --git a/apps/worker/test-dist-q/entities/query.js b/apps/worker/test-dist-q/entities/query.js new file mode 100644 index 00000000..79f93c4f --- /dev/null +++ b/apps/worker/test-dist-q/entities/query.js @@ -0,0 +1,154 @@ +// High-level read helpers. `loadEntity` returns the canonical envelope +// every consumer can rely on; `searchEntities` is a thin filter DSL that +// joins entity_summary + entity_tags for sub-50ms list responses. +import { getEffectiveFacts, loadCurrentOverrides } from "./facts"; +export async function loadEntity(env, id, opts) { + const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first(); + if (!ent) + return null; + // Task #3 (Editable Profiles): use the SAME shared resolver as the + // summary rebuilder so the two read sites cannot drift. The resolver + // returns one EffectiveFact list with canonical rows + dethroned + // attempts marked overridden_attempt=true; we split that into + // facts[] (canonical) and attempts[] (diff strip). + const [roles, effective, channels, tags, summary, overridesMap] = await Promise.all([ + env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(id).all(), + getEffectiveFacts(env, id, { includeNonCurrent: !!opts?.includeNonCurrent, limit: 500 }), + env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified, is_dnc FROM channels WHERE entity_id = ?`).bind(id).all(), + env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(id).all(), + env.DB.prepare(`SELECT * FROM entity_summary WHERE entity_id = ?`).bind(id).first(), + loadCurrentOverrides(env, id), + ]); + const factRows = []; + const attempts = []; + for (const e of effective) { + const row = { + id: e.id, predicate: e.predicate, + value_text: e.value_text, value_number: e.value_number, + value_json: e.value_json, value_entity_id: e.value_entity_id, + source: e.source, source_kind: e.source_kind, + confidence: e.confidence, verified_score: e.verified_score, + observed_at: e.observed_at, is_current: e.is_current, + superseded_by_override: e.superseded_by_override, + }; + if (e.overridden_attempt) + attempts.push(row); + else + factRows.push(row); + } + const overrideArr = Array.from(overridesMap.values()).map((ov) => ({ + id: ov.id, + predicate: ov.predicate, + value_text: ov.value_text, + value_number: ov.value_numeric, + value_json: ov.value_json ? (() => { try { + return JSON.parse(ov.value_json); + } + catch { + return ov.value_json; + } })() : null, + overridden_at: ov.overridden_at, + })); + return { + id, kind: ent.kind, entity: ent, + roles: (roles.results ?? []), + facts: factRows, + attempts, + channels: (channels.results ?? []), + tags: (tags.results ?? []), + summary: summary ?? null, + overrides: overrideArr, + }; +} +export async function searchEntities(env, f) { + // IMPORTANT: bind order must match SQL placeholder order. Since the + // tag JOINs appear *before* the WHERE clause in the final SQL, their + // binds must be pushed first. We build two separate bind arrays and + // concatenate them in SQL order at the end. + const joinBinds = []; + const whereBinds = []; + const where = ["s.status = 'active'"]; + if (f.kind) { + where.push("s.kind = ?"); + whereBinds.push(f.kind); + } + if (f.role) { + where.push("s.primary_role = ?"); + whereBinds.push(f.role); + } + if (f.country_iso2) { + where.push("s.country_iso2 = ?"); + whereBinds.push(f.country_iso2.toUpperCase()); + } + if (typeof f.check_min_usd === "number") { + where.push("s.check_size_max_usd >= ?"); + whereBinds.push(f.check_min_usd); + } + if (typeof f.check_max_usd === "number") { + where.push("s.check_size_min_usd <= ?"); + whereBinds.push(f.check_max_usd); + } + if (f.has_unicorn) { + where.push("s.unicorn_count > 0"); + } + if (typeof f.min_fit === "number") { + where.push("s.fit_max_score >= ?"); + whereBinds.push(f.min_fit); + } + if (typeof f.min_intent === "number") { + where.push("s.intent_score >= ?"); + whereBinds.push(f.min_intent); + } + if (f.q) { + where.push("(lower(s.display_name) LIKE ? OR lower(s.primary_domain) LIKE ? OR lower(s.primary_email) LIKE ?)"); + const q = `%${f.q.toLowerCase()}%`; + whereBinds.push(q, q, q); + } + // Tag filters require a JOIN per taxonomy so we can intersect. + // NOTE: `has_role` is *not* a tag — roles live in entity_roles + // (addRole writes there, not entity_tags). The previous JOIN on + // entity_tags taxonomy='role' returned empty results for every + // wrapper helper (listFirms, listInvestors, listCompanies, + // listAccounts, listBuyers, listFounders). We now JOIN entity_roles + // directly for role membership. + const joins = []; + let tagJoinIdx = 0; + for (const [tax, slug] of [["sector", f.sector], ["stage", f.stage], ["geo", f.geo]]) { + if (!slug) + continue; + const alias = `t${++tagJoinIdx}`; + joins.push(`JOIN entity_tags ${alias} ON ${alias}.entity_id = s.entity_id AND ${alias}.taxonomy = ? AND ${alias}.slug = ?`); + joinBinds.push(tax, slug); + } + if (f.has_role) { + joins.push(`JOIN entity_roles er ON er.entity_id = s.entity_id AND er.role = ?`); + joinBinds.push(f.has_role); + } + const sortCol = (() => { + switch (f.sort) { + case "intent": return "s.intent_score DESC"; + case "quality": return "s.quality_score DESC"; + case "updated": return "s.rebuilt_at DESC"; + default: return "s.fit_max_score DESC"; + } + })(); + const limit = Math.min(Math.max(1, f.limit ?? 50), 200); + const offset = Math.max(0, f.offset ?? 0); + const sql = `SELECT s.* FROM entity_summary s ${joins.join(" ")} + WHERE ${where.join(" AND ")} + ORDER BY ${sortCol}, s.entity_id ASC + LIMIT ? OFFSET ?`; + const r = await env.DB.prepare(sql).bind(...joinBinds, ...whereBinds, limit + 1, offset).all(); + const rows = r.results ?? []; + const hasMore = rows.length > limit; + return { + items: (hasMore ? rows.slice(0, limit) : rows), + next_offset: hasMore ? offset + limit : null, + }; +} +export const listFirms = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "firm" }); +export const listInvestors = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "investor" }); +export const listCompanies = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: f.has_role ?? "company" }); +export const listAccounts = (env, f = {}) => searchEntities(env, { ...f, kind: "org", has_role: "account" }); +export const listBuyers = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "buyer" }); +export const listFounders = (env, f = {}) => searchEntities(env, { ...f, kind: "person", has_role: "founder" }); diff --git a/apps/worker/test-dist-q/entities/roles.js b/apps/worker/test-dist-q/entities/roles.js new file mode 100644 index 00000000..2b16577f --- /dev/null +++ b/apps/worker/test-dist-q/entities/roles.js @@ -0,0 +1,120 @@ +import { isGarbage, logDataQuality, classifyPersonName } from "./garbage"; +export async function createEntity(env, init) { + // Task #9: pre-insert garbage guard. The pure heuristic detector + // (no AI call here — keep createEntity synchronous and cheap on the + // hot write path) rejects HTML page titles / nav strings / UI + // labels that the crawler may have mistaken for entity names. + // Returns null + audit row instead of throwing so callers can + // skip without crashing the broader import. The AI second opinion + // runs only in the cron sweep, NOT inline on every write. + const verdict = isGarbage({ + kind: init.kind, + display_name: init.display_name ?? null, + primary_url: init.primary_url ?? null, + primary_domain: init.primary_domain ?? null, + primary_email_key: init.primary_email_key ?? null, + primary_linkedin_key: init.primary_linkedin_key ?? null, + }); + if (verdict.is_garbage) { + console.log("garbage.pre_insert_rejected", JSON.stringify({ + kind: init.kind, display_name: init.display_name, reasons: verdict.reasons, + })); + // Log to data_quality_log with a synthetic entity_id so operators + // can audit rejected writes too. We use a `rejected:` prefix to + // distinguish from soft-deleted entities (which carry real ids). + void logDataQuality(env, "rejected:" + (init.display_name ?? "").slice(0, 100), "pre_insert_rejected", verdict.reasons, "pre_insert_guard", null).catch(() => undefined); + return null; + } + // Task #6: reclassify-on-write. A `person` whose display name is clearly + // an organization ("Intel Capital", "Mendoza Ventures") is written as an + // `org` so it never lands in the People list in the first place. A strong + // personal identifier (personal LinkedIn /in/ or email) contradicting the + // org-suffix name suppresses the flip — never mislabel a likely real + // person. Junk names were already rejected by the isGarbage guard above. + let effectiveKind = init.kind; + let reclassifiedOrgRole = null; + if (init.kind === "person") { + const cls = classifyPersonName(init.display_name ?? null); + if (cls.verdict === "organization" && cls.orgRole) { + const personalLinkedin = !!init.primary_linkedin_key && /(^|\/)in\//i.test(init.primary_linkedin_key); + const hasEmail = !!init.primary_email_key; + if (!personalLinkedin && !hasEmail) { + effectiveKind = "org"; + reclassifiedOrgRole = cls.orgRole; + } + } + } + const id = crypto.randomUUID(); + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO u_entities ( + id, kind, display_name, primary_url, primary_domain, + primary_email_key, primary_linkedin_key, primary_twitter_handle, primary_github_handle, + status, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'active', ?, ?)`).bind(id, effectiveKind, init.display_name ?? null, init.primary_url ?? null, init.primary_domain ?? null, init.primary_email_key ?? null, init.primary_linkedin_key ?? null, init.primary_twitter_handle ?? null, init.primary_github_handle ?? null, now, now).run(); + await env.DB.prepare(`INSERT INTO entity_history (id, entity_id, action, source, changed_at) VALUES (?, ?, 'create', 'system', ?)`).bind(crypto.randomUUID(), id, now).run(); + // Task #8: enqueue persona ↔ entity matching for newly-created + // person entities so a freshly-created founder/operator appears in + // matching personas' candidate lists within minutes — even before + // any career/title facts are written. KV-debounced inside trigger. + if (effectiveKind === "person") { + try { + const { triggerEntityMatchRefresh } = await import("../services/personaMatchTrigger.js"); + void triggerEntityMatchRefresh(env, id).catch(() => undefined); + } + catch { /* best-effort */ } + } + // Task #6: stamp the inferred org role + an audit row when a person was + // reclassified to an org on write, so the operator console can trace it. + if (reclassifiedOrgRole) { + await addRole(env, id, reclassifiedOrgRole, { is_primary: true, source: "garbage_reclassify_on_write" }); + void logDataQuality(env, id, "reclassified", [`org_role:${reclassifiedOrgRole}`, "reclassified_on_write"], "pre_insert_guard", null).catch(() => undefined); + } + // Task #3 (AI Profile Filler): auto-trigger a profile fill for newly + // created org entities that have a website but no facts yet (the + // signal-poor "low confidence" case the spec calls out). Dispatched + // via WF binding when available so the cost is async and respects + // the daily neuron cap. No-op when the binding isn't configured. + if (effectiveKind === "org" && (init.primary_url || init.primary_domain) && !init.suppressAutoProfileFill) { + const wf = env.WF_PROFILE_FILLER; + if (wf) { + try { + void wf.create({ params: { entityId: id, force: false, triggeredBy: "auto:entity_created" } }).catch(() => undefined); + } + catch { /* best-effort */ } + } + } + // Task #4 (Relationship Inference Worker): debounced enqueue into + // relationship_infer_queue (migration 377). The consolidated nightly + // slot drains the queue with the per-entity orchestrator pass. + try { + const { enqueueRelInfer } = await import("../services/relationships/orchestrator.js"); + void enqueueRelInfer(env, id, `created:${effectiveKind}`).catch(() => undefined); + } + catch { /* best-effort */ } + return (await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(id).first()); +} +export async function addRole(env, entityId, role, opts) { + try { + await env.DB.prepare(`INSERT INTO entity_roles (entity_id, role, is_primary, source, confidence) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, role) DO UPDATE SET + is_primary = MAX(is_primary, excluded.is_primary), + confidence = MAX(confidence, excluded.confidence)`).bind(entityId, role, opts?.is_primary ? 1 : 0, opts?.source ?? null, opts?.confidence ?? 1).run(); + } + catch (e) { + console.warn("addRole failed", role, e.message); + } +} +export async function getLegacyEntityId(env, table, legacyId) { + const r = await env.DB.prepare(`SELECT entity_id FROM entity_legacy_map WHERE legacy_table = ? AND legacy_id = ?`).bind(table, String(legacyId)).first(); + return r?.entity_id ?? null; +} +export async function setLegacyEntityId(env, table, legacyId, entityId) { + try { + await env.DB.prepare(`INSERT INTO entity_legacy_map (legacy_table, legacy_id, entity_id) + VALUES (?, ?, ?) ON CONFLICT DO NOTHING`).bind(table, String(legacyId), entityId).run(); + } + catch (e) { + console.warn("setLegacyEntityId failed", table, legacyId, e.message); + } +} diff --git a/apps/worker/test-dist-q/entities/summary.js b/apps/worker/test-dist-q/entities/summary.js new file mode 100644 index 00000000..fc6eef0b --- /dev/null +++ b/apps/worker/test-dist-q/entities/summary.js @@ -0,0 +1,121 @@ +// Rebuild `entity_summary` for one entity from its current facts + +// channels + tags + roles. Runs inside the queue consumer. +import { getEffectiveFacts } from "./facts"; +const SOURCE_PRIORITY = { + manual: 5, enrichment: 4, import: 3, scrape: 2, ai: 1, inferred: 0, +}; +function pickBestFact(rows) { + if (!rows.length) + return null; + return rows.slice().sort((a, b) => { + const sa = (a.confidence ?? 0) * 100 + (SOURCE_PRIORITY[a.source_kind] ?? 0) * 10 + Date.parse(a.observed_at) / 1e12; + const sb = (b.confidence ?? 0) * 100 + (SOURCE_PRIORITY[b.source_kind] ?? 0) * 10 + Date.parse(b.observed_at) / 1e12; + return sb - sa; + })[0]; +} +function txt(rows, predicate) { + const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); + return f?.value_text ?? null; +} +function num(rows, predicate) { + const f = pickBestFact(rows.filter((r) => r.predicate === predicate)); + return f?.value_number ?? null; +} +export async function rebuildSummary(env, entityId) { + const ent = await env.DB.prepare(`SELECT * FROM u_entities WHERE id = ?`).bind(entityId).first(); + if (!ent) + return false; + if (ent.status === "merged" || ent.status === "soft_deleted") { + await env.DB.prepare(`DELETE FROM entity_summary WHERE entity_id = ?`).bind(entityId).run(); + return true; + } + // Task #3 (Editable Profiles): the summary input is the EFFECTIVE + // facts view — overrides win, overridden_attempt rows are filtered. + // Same resolver as the per-entity read path in query.ts, so the two + // call sites cannot drift. + const [effective, tagsRes, rolesRes, channelsRes] = await Promise.all([ + getEffectiveFacts(env, entityId), + env.DB.prepare(`SELECT taxonomy, slug, weight FROM entity_tags WHERE entity_id = ?`).bind(entityId).all(), + env.DB.prepare(`SELECT role, is_primary, confidence FROM entity_roles WHERE entity_id = ?`).bind(entityId).all(), + env.DB.prepare(`SELECT kind, canonical, display, is_primary, is_verified FROM channels WHERE entity_id = ?`).bind(entityId).all(), + ]); + const facts = effective + .filter((e) => !e.overridden_attempt) + .map((e) => ({ + predicate: e.predicate, + value_text: e.value_text, + value_number: e.value_number, + value_json: e.value_json != null ? (typeof e.value_json === "string" ? e.value_json : JSON.stringify(e.value_json)) : null, + value_entity_id: e.value_entity_id, + confidence: e.confidence, + observed_at: e.observed_at, + source_kind: e.source_kind, + })); + const tags = tagsRes.results ?? []; + const roles = rolesRes.results ?? []; + const channels = channelsRes.results ?? []; + const primaryRole = (roles.find((r) => r.is_primary === 1) ?? roles[0])?.role ?? null; + const display = ent.display_name ?? txt(facts, "name") ?? txt(facts, "display_name"); + const country = txt(facts, "country_iso2"); + const region = txt(facts, "region"); + const city = txt(facts, "city"); + const sectors = tags.filter((t) => t.taxonomy === "sector").map((t) => t.slug); + const stages = tags.filter((t) => t.taxonomy === "stage").map((t) => t.slug); + const geos = tags.filter((t) => t.taxonomy === "geo").map((t) => t.slug); + const checkMin = num(facts, "check_size_min_usd"); + const checkMax = num(facts, "check_size_max_usd"); + const fitMax = num(facts, "fit_max_score") ?? 0; + const intent = num(facts, "intent_score") ?? 0; + const unicornCount = Math.round(num(facts, "unicorn_count") ?? 0); + const primaryEmailChan = channels.filter((c) => c.kind === "email").sort((a, b) => (b.is_primary - a.is_primary) || (b.is_verified - a.is_verified))[0]; + const primaryLinkedinChan = channels.filter((c) => c.kind === "linkedin")[0]; + const employerEntityId = (facts.find((f) => f.predicate === "employer" && f.value_entity_id) ?? null)?.value_entity_id ?? null; + let employerDisplay = null; + if (employerEntityId) { + const r = await env.DB.prepare(`SELECT display_name FROM u_entities WHERE id = ?`).bind(employerEntityId).first(); + employerDisplay = r?.display_name ?? null; + } + else { + employerDisplay = txt(facts, "primary_employer") ?? txt(facts, "org"); + } + // Quality: coverage × confidence average × source diversity, scaled 0..100. + const coverage = Math.min(1, facts.length / 12); + const confAvg = facts.length ? facts.reduce((s, f) => s + (f.confidence ?? 0), 0) / facts.length : 0; + const diversity = Math.min(1, new Set(facts.map((f) => f.source_kind)).size / 3); + const quality = Math.round((coverage * 0.5 + confAvg * 0.3 + diversity * 0.2) * 100); + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO entity_summary ( + entity_id, kind, display_name, primary_role, primary_employer, primary_employer_entity_id, + country_iso2, region, city, sectors_csv, stages_csv, geos_csv, + check_size_min_usd, check_size_max_usd, primary_email, primary_linkedin, + primary_domain, quality_score, fit_max_score, intent_score, unicorn_count, status, rebuilt_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(entity_id) DO UPDATE SET + kind = excluded.kind, + display_name = excluded.display_name, + primary_role = excluded.primary_role, + primary_employer = excluded.primary_employer, + primary_employer_entity_id = excluded.primary_employer_entity_id, + country_iso2 = excluded.country_iso2, + region = excluded.region, + city = excluded.city, + sectors_csv = excluded.sectors_csv, + stages_csv = excluded.stages_csv, + geos_csv = excluded.geos_csv, + check_size_min_usd = excluded.check_size_min_usd, + check_size_max_usd = excluded.check_size_max_usd, + primary_email = excluded.primary_email, + primary_linkedin = excluded.primary_linkedin, + primary_domain = excluded.primary_domain, + quality_score = excluded.quality_score, + fit_max_score = excluded.fit_max_score, + intent_score = excluded.intent_score, + unicorn_count = excluded.unicorn_count, + status = excluded.status, + rebuilt_at = excluded.rebuilt_at`).bind(entityId, ent.kind, display, primaryRole, employerDisplay, employerEntityId, country, region, city, sectors.join(","), stages.join(","), geos.join(","), checkMin != null ? Math.round(checkMin) : null, checkMax != null ? Math.round(checkMax) : null, primaryEmailChan?.canonical ?? null, primaryLinkedinChan?.canonical ?? null, ent.primary_domain, quality, fitMax, intent, unicornCount, ent.status, now).run(); + // Update u_entities.quality_score + last_summary_at so list paths that + // still read from `u_entities` see the latest score. + await env.DB.prepare(`UPDATE u_entities SET quality_score = ?, last_summary_at = ?, updated_at = ? WHERE id = ?`) + .bind(quality, now, now, entityId).run(); + return true; +} diff --git a/apps/worker/test-dist-q/entities/summaryQueue.js b/apps/worker/test-dist-q/entities/summaryQueue.js new file mode 100644 index 00000000..ccbf3c4b --- /dev/null +++ b/apps/worker/test-dist-q/entities/summaryQueue.js @@ -0,0 +1,26 @@ +// Enqueue + consume `rebuild_summary` work. Uses the existing LEAD_QUEUE +// to avoid provisioning a second queue; the consumer in index.ts +// dispatches by message shape. +import { rebuildSummary } from "./summary"; +export function isRebuildSummaryMessage(m) { + return !!m && typeof m === "object" && m.type === "rebuild_summary" + && typeof m.entityId === "string"; +} +// We intentionally do *not* debounce via KV: the previous design dropped +// later writes when a debounce key was already set, which left +// entity_summary stale until the next mutation. The queue handler can +// coalesce duplicates at consume time if needed; in the meantime, the +// extra work is one upsert into a tiny rollup table. +export async function enqueueSummaryRebuild(env, entityId) { + if (!entityId) + return; + try { + await env.LEAD_QUEUE.send({ type: "rebuild_summary", entityId }); + } + catch (e) { + console.warn("enqueueSummaryRebuild failed", entityId, e.message); + } +} +export async function handleSummaryMessage(env, m) { + await rebuildSummary(env, m.entityId); +} diff --git a/apps/worker/test-dist-q/entities/tags.js b/apps/worker/test-dist-q/entities/tags.js new file mode 100644 index 00000000..a28e6fe3 --- /dev/null +++ b/apps/worker/test-dist-q/entities/tags.js @@ -0,0 +1,37 @@ +export async function addTag(env, t) { + if (!t.entity_id || !t.slug) + return; + const slug = String(t.slug).trim().toLowerCase(); + if (!slug) + return; + try { + await env.DB.prepare(`INSERT INTO entity_tags (entity_id, taxonomy, slug, weight, source) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(entity_id, taxonomy, slug) + DO UPDATE SET weight = MAX(weight, excluded.weight)`).bind(t.entity_id, t.taxonomy, slug, t.weight ?? 1, t.source ?? null).run(); + } + catch (e) { + console.warn("addTag failed", t.taxonomy, slug, e.message); + } +} +export async function addTagsFromJsonArray(env, entityId, taxonomy, rawJsonArray, source) { + if (!rawJsonArray) + return 0; + let arr; + try { + arr = JSON.parse(rawJsonArray); + } + catch { + return 0; + } + if (!Array.isArray(arr)) + return 0; + let n = 0; + for (const v of arr) { + if (typeof v !== "string") + continue; + await addTag(env, { entity_id: entityId, taxonomy, slug: v, source }); + n += 1; + } + return n; +} diff --git a/apps/worker/test-dist-q/errors.js b/apps/worker/test-dist-q/errors.js new file mode 100644 index 00000000..653c1b1a --- /dev/null +++ b/apps/worker/test-dist-q/errors.js @@ -0,0 +1,328 @@ +// Centralized error taxonomy for the worker (Task #27). +// +// Every operational failure should be either: +// 1. an `AppError` subclass thrown explicitly, OR +// 2. caught and re-thrown via `wrapUnknown(e, code, ctx)` so the global +// onError handler in `index.ts` can serialize it to JSON, log it to +// `error_log`, mirror it to Analytics Engine, and attach a request_id. +// +// All AppErrors carry: +// - code: stable machine string from the ErrCode union below. +// - status: HTTP status to return when surfaced via API. +// - kind: high-level category for UI grouping. +// - retryable: hint to the queue/retry layer. +// - context: free-form structured context (job_id, url, host, provider…). +export class AppError extends Error { + code; + kind; + status; + retryable; + context; + cause; + constructor(opts) { + super(opts.message ?? opts.code); + this.name = "AppError"; + this.code = opts.code; + this.kind = opts.kind; + this.status = opts.status ?? defaultStatusForKind(opts.kind); + this.retryable = opts.retryable ?? defaultRetryableForKind(opts.kind); + this.context = opts.context ?? {}; + if (opts.cause instanceof Error) + this.cause = opts.cause; + } + toJSON(requestId) { + const out = { + error: this.code, + code: this.code, + kind: this.kind, + status: this.status, + message: this.message, + retryable: this.retryable, + }; + if (Object.keys(this.context).length) + out.context = this.context; + if (requestId) + out.request_id = requestId; + if (this.cause) { + const c = { + name: this.cause.name, + message: this.cause.message, + }; + if (this.cause.stack) + c.stack = this.cause.stack; + out.cause = c; + } + return out; + } +} +function defaultStatusForKind(kind) { + switch (kind) { + case "validation": return 400; + case "auth": return 401; + case "permanent": return 422; + case "config": return 500; + case "upstream": return 502; + case "transient": return 503; + // Task #72: a benign skip is not an error surface — neutral 200. + case "skip": return 200; + case "internal": + default: return 500; + } +} +function defaultRetryableForKind(kind) { + return kind === "transient" || kind === "upstream"; +} +// ---- Common subclasses (sugar; AppError directly is also fine) ---------- +export class ValidationError extends AppError { + constructor(code, message, context) { + super({ code, kind: "validation", status: 400, message, retryable: false, ...(context ? { context } : {}) }); + this.name = "ValidationError"; + } +} +export class NotFoundError extends AppError { + constructor(resource, id) { + super({ + code: "not_found", + kind: "permanent", + status: 404, + message: `${resource}${id ? ` ${id}` : ""} not found`, + retryable: false, + context: id ? { resource, id } : { resource }, + }); + this.name = "NotFoundError"; + } +} +export class AuthError extends AppError { + constructor(code, message, context) { + super({ + code, + kind: "auth", + status: code === "forbidden" ? 403 : 401, + message: message ?? code, + retryable: false, + ...(context ? { context } : {}), + }); + this.name = "AuthError"; + } +} +export class UpstreamError extends AppError { + constructor(provider, message, context) { + // Constructed at runtime; the closed ErrCode enum already enumerates + // every known provider so this assertion is the only escape. + const code = `upstream_${provider}`; + super({ + code, + kind: "upstream", + status: 502, + message, + retryable: true, + context: { provider, ...(context ?? {}) }, + }); + this.name = "UpstreamError"; + } +} +export class ScrapeBlockedError extends AppError { + constructor(host, reason, context) { + super({ + code: "scrape_blocked", + kind: "permanent", + status: 403, + message: `${host}: ${reason}`, + retryable: false, + context: { host, reason, ...(context ?? {}) }, + }); + this.name = "ScrapeBlockedError"; + } +} +export class BudgetExhaustedError extends AppError { + constructor(scope, context) { + super({ + code: "budget_exhausted", + kind: "permanent", + status: 429, + message: `Budget exhausted for ${scope}`, + retryable: false, + context: { scope, ...(context ?? {}) }, + }); + this.name = "BudgetExhaustedError"; + } +} +export class TransientError extends AppError { + constructor(code, message, context) { + super({ code, kind: "transient", status: 503, message, retryable: true, ...(context ? { context } : {}) }); + this.name = "TransientError"; + } +} +// ---- Wrappers -------------------------------------------------------------- +/** Convert any unknown thrown value into an AppError (idempotent). */ +export function wrapUnknown(e, code, context) { + if (e instanceof AppError) { + if (context) + Object.assign(e.context, context); + return e; + } + const err = e instanceof Error ? e : new Error(typeof e === "string" ? e : safeJson(e)); + // Heuristic upgrade: detect common transient patterns from the cause. + const guessed = classify(err); + const opts = { + code: guessed?.code ?? code, + kind: guessed?.kind ?? "internal", + message: err.message || code, + cause: err, + }; + if (guessed) + opts.retryable = guessed.retryable; + if (context) + opts.context = context; + return new AppError(opts); +} +/** Type guard. */ +export function isAppError(e) { + return e instanceof AppError; +} +/** + * Heuristic classifier for stringly-typed errors thrown by the existing + * codebase or by the platform (D1, fetch, Workers AI). Returns null if no + * pattern matches; callers should then use the supplied default code/kind. + * + * Required by Task #27 acceptance: every logged failure has a well-typed + * code, even when thrown deep in legacy code that hasn't migrated to + * AppError yet. + */ +export function classify(err) { + if (err instanceof AppError) + return { code: err.code, kind: err.kind, retryable: err.retryable }; + const msg = (err instanceof Error ? err.message : String(err ?? "")).toLowerCase(); + if (!msg) + return null; + // Task #70: Cloudflare's per-invocation subrequest cap surfaces as + // "Too many subrequests by single Worker invocation", which bubbles up + // wrapped in a `fetch_failed:proxy_error:...` (or `fetch_error:...`) + // string. This is NOT a permanent scrape block — the page is fine, the + // invocation just ran out of budget — so it must be classified + // transient/retryable AHEAD of the generic `fetch_failed:` permanent + // rule below. Matched specifically (not all proxy_errors) so genuine + // upstream proxy failures still dead-letter as before. + // + // `subrequest_budget_exhausted` is our OWN pre-emptive refusal (the + // crawl-path budget stopped a fetch before it could trip the platform + // cap); it is the same condition and must retry identically. + if (msg.includes("too many subrequests") || msg.includes("subrequest_budget")) { + return { code: "subrequest_limit", kind: "transient", retryable: true }; + } + // Pipeline-level fetch/scrape sentinels. These reasons are bubbled up + // from the scraper as plain `Error("fetch_failed::status=")` + // (see scraper/pipeline.ts). They are expected operational outcomes — + // not real internal errors — so we map them to typed codes the queue + // can dead-letter without paging. + if (msg.includes("scraping_api_not_configured") || + msg.includes("proxy_not_configured") || + msg.includes("browser_binding_unavailable") || + msg.includes("puppeteer_module_missing")) { + return { code: "config_missing", kind: "config", retryable: false }; + } + // Task #72: robots.txt / ToS blocks are EXPECTED, benign policy outcomes, + // not internal errors — honoring a host's robots.txt is correct behavior. + // Classify them as the `skip` kind so the queue routes them to the `skipped` + // terminal status (no error_log row, never retried) instead of failing / + // dead-lettering them as scrape errors. NB: the scraper emits the token + // `robots_disallow` (see scraper/robots.ts); the old code only matched the + // `robots_disallowed` spelling and silently fell through to the generic + // fetch_failed → permanent rule, so the block surfaced as a red 422. + if (msg.includes("robots_disallow") || msg.includes("tos_blocked")) { + const code = msg.includes("tos_blocked") ? "tos_blocked" : "robots_disallowed"; + return { code, kind: "skip", retryable: false }; + } + // Gated sources still need an operator manual-paste; that's a permanent + // scrape block (the queue preflight already skips it earlier — this is the + // fetcher backstop, kept permanent so its behavior is unchanged). + if (msg.includes("gated_source_use_manual_paste")) { + return { code: "scrape_blocked", kind: "permanent", retryable: false }; + } + if (msg.includes("no_table_found")) { + return { code: "parse_error", kind: "validation", retryable: false }; + } + if (msg.startsWith("fetch_failed:") || msg.includes(":fetch_failed:")) { + // Task #71: a `fetch_failed:` message can carry an embedded upstream HTTP + // status (e.g. "fetch_failed:status_429:status=429"). A 429 rate-limit and + // any 5xx are TRANSIENT — the page is fine, the upstream is just briefly + // refusing — so they must retry with backoff, not get dropped as a + // permanent scrape block on attempt 1. Parse the embedded status BEFORE the + // generic permanent fallback. A genuine 4xx (403/404/...) or a + // fetch_failed with no recoverable status still resolves to permanent + // scrape_blocked (prior behavior preserved). The dedicated scrape sentinels + // (robots/tos/gated/config) matched above stay permanent regardless. + const embedded = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); + if (embedded) { + const s = Number(embedded[1]); + if (s === 429) + return { code: "rate_limited", kind: "transient", retryable: true }; + if (s >= 500) + return { code: "fetch.http_5xx", kind: "transient", retryable: true }; + } + // Generic fetch_failed without a recoverable status → upstream/permanent. + return { code: "scrape_blocked", kind: "permanent", retryable: false }; + } + // Network / fetch. + if (msg.includes("aborted") || msg.includes("timeout") || msg.includes("timed out")) { + return { code: "fetch.timeout", kind: "transient", retryable: true }; + } + if (msg.includes("network connection lost") || msg.includes("econnreset") || msg.includes("ehostunreach")) { + return { code: "fetch.error", kind: "transient", retryable: true }; + } + // HTTP status codes embedded in the message (e.g. "status_503", "status=404", + // "status: 502", or a bare " 404 " token). + const statusMatch = msg.match(/status[_=: ]\s*(\d{3})/) ?? msg.match(/\b(4\d{2}|5\d{2})\b/); + if (statusMatch) { + const s = Number(statusMatch[1]); + if (s === 429) + return { code: "rate_limited", kind: "transient", retryable: true }; + if (s >= 500) + return { code: "fetch.http_5xx", kind: "transient", retryable: true }; + if (s >= 400) + return { code: "fetch.http_4xx", kind: "permanent", retryable: false }; + } + // D1 / Vectorize / Workers AI. + if (msg.includes("d1_error") || msg.includes("sqlite_") || msg.includes("database is locked")) { + return { code: "db_error", kind: "transient", retryable: true }; + } + if (msg.includes("vectorize")) + return { code: "vectorize_error", kind: "upstream", retryable: true }; + if (msg.includes("ai.run") || msg.includes("workers ai")) + return { code: "ai_error", kind: "upstream", retryable: true }; + // Parsing. + if (msg.includes("unexpected token") || msg.includes("json")) + return { code: "json_parse_error", kind: "validation", retryable: false }; + if (msg.includes("invalid url") || msg.includes("uri malformed")) + return { code: "validation_failed", kind: "validation", retryable: false }; + // Auth. + if (msg.includes("jwt") || msg.includes("unauthorized") || msg.includes("forbidden")) { + return { code: "unauthorized", kind: "auth", retryable: false }; + } + return null; +} +/** + * Task #72: is this thrown value a benign policy skip (robots.txt / ToS) + * rather than a real fetch failure? Benign skips end a job in the `skipped` + * terminal status (no error_log, no retry), NOT failed/dead_letter. Returns + * the stable `skip_code` + a human reason, or null for everything else (which + * the caller then classifies / retries / dead-letters normally). + */ +export function isBenignSkip(err) { + const cls = err instanceof AppError + ? { code: err.code, kind: err.kind } + : classify(err); + if (!cls || cls.kind !== "skip") + return null; + const skip_code = cls.code === "tos_blocked" ? "tos_blocked" : "robots_disallow"; + const reason = err instanceof Error ? err.message : String(err ?? skip_code); + return { skip_code, reason }; +} +function safeJson(v) { + try { + return JSON.stringify(v); + } + catch { + return String(v); + } +} diff --git a/apps/worker/test-dist-q/personas/repo.js b/apps/worker/test-dist-q/personas/repo.js new file mode 100644 index 00000000..e0170927 --- /dev/null +++ b/apps/worker/test-dist-q/personas/repo.js @@ -0,0 +1,403 @@ +// Task #46: persona data-access layer. +export const PERSONA_FIELDS = [ + "name", "kind", "status", "thesis", + "hard_filters_json", "size_min", "size_max", "size_bands_json", + "geos_json", "industries_json", + "techs_required_json", "techs_preferred_json", "techs_excluded_json", + "signal_kinds_json", "buyer_titles_json", "buyer_seniority_json", "buyer_departments_json", + "weights_json", "semantic_fit_threshold", "recency_boost", +]; +function parseJsonArr(s) { + if (!s) + return []; + try { + const v = JSON.parse(s); + return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; + } + catch { + return []; + } +} +function parseJsonObj(s) { + if (!s) + return {}; + try { + const v = JSON.parse(s); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +export function rowToSpec(row) { + const w = parseJsonObj(row.weights_json); + // Legacy scorer only understands account/buyer. New taxonomy kinds + // fall through to "buyer" for spec purposes (the kind dispatcher in + // services/personas/kinds owns real matching for new kinds; this + // spec is only used by the legacy persona_matches code path). + const legacyKind = row.kind === "account" || row.kind === "account_company" ? "account" : "buyer"; + return { + id: row.id, + kind: legacyKind, + size_min: row.size_min, + size_max: row.size_max, + size_bands: parseJsonArr(row.size_bands_json), + geos: parseJsonArr(row.geos_json), + industries: parseJsonArr(row.industries_json), + techs_required: parseJsonArr(row.techs_required_json), + techs_preferred: parseJsonArr(row.techs_preferred_json), + techs_excluded: parseJsonArr(row.techs_excluded_json), + signal_kinds: parseJsonArr(row.signal_kinds_json), + buyer_titles: parseJsonArr(row.buyer_titles_json), + buyer_seniority: parseJsonArr(row.buyer_seniority_json), + buyer_departments: parseJsonArr(row.buyer_departments_json), + hard_filters: parseJsonObj(row.hard_filters_json), + weights: w, + semantic_fit_threshold: row.semantic_fit_threshold ?? 0.55, + recency_boost: row.recency_boost ?? 0, + }; +} +export async function listPersonas(env, opts) { + const status = opts?.status ?? "active"; + const limit = Math.min(Math.max(1, opts?.limit ?? 200), 500); + const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = ? ORDER BY last_modified DESC LIMIT ?`).bind(status, limit).all(); + return r.results ?? []; +} +export async function getPersona(env, id) { + const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ? AND deleted_at IS NULL`).bind(id).first(); + return r ?? null; +} +export async function getPersonaIncludingDeleted(env, id) { + const r = await env.DB.prepare(`SELECT * FROM personas WHERE id = ?`).bind(id).first(); + return r ?? null; +} +export async function insertPersona(env, body, by, idOverride) { + const id = idOverride ?? crypto.randomUUID(); + const now = new Date().toISOString(); + const cols = ["id", "created_by", "created_at", "updated_at", "last_modified", ...PERSONA_FIELDS]; + const binds = [id, by ?? null, now, now, now]; + // Defaults for NOT NULL columns when the caller omits them (e.g. the + // seed loader). Without this, the seed path threw + // `NOT NULL constraint failed: personas.status` and surfaced as a + // db_error on the Personas page. + const defaults = { status: "active", kind: "account" }; + for (const f of PERSONA_FIELDS) { + const v = body[f]; + binds.push(v ?? defaults[f] ?? null); + } + await env.DB.prepare(`INSERT INTO personas (${cols.join(",")}) VALUES (${cols.map(() => "?").join(",")})`).bind(...binds).run(); + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'created', ?, ?)`) + .bind(crypto.randomUUID(), id, body.name, by ?? null).run(); + const row = await getPersona(env, id); + return row; +} +export async function updatePersona(env, id, patch, by) { + const cur = await getPersona(env, id); + if (!cur) + return null; + const allowed = new Set(PERSONA_FIELDS); + const sets = []; + const binds = []; + const hist = []; + for (const [k, v] of Object.entries(patch)) { + if (!allowed.has(k)) + continue; + sets.push(`${k} = ?`); + binds.push(v); + const before = cur[k]; + if (before !== v) + hist.push({ field: k, old: before, nw: v }); + } + if (!sets.length) + return cur; + const now = new Date().toISOString(); + binds.push(now, now, id); + await env.DB.prepare(`UPDATE personas SET ${sets.join(", ")}, updated_at = ?, last_modified = ? WHERE id = ?`).bind(...binds).run(); + for (const h of hist) { + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, old_value, new_value, changed_by) VALUES (?, ?, ?, ?, ?, ?)`) + .bind(crypto.randomUUID(), id, h.field, h.old != null ? String(h.old) : null, h.nw != null ? String(h.nw) : null, by ?? null).run(); + } + return await getPersona(env, id); +} +export async function setPersonaEmbeddingMeta(env, id, dim, text) { + const now = new Date().toISOString(); + await env.DB.prepare(`UPDATE personas SET embedding_dim = ?, embedded_at = ?, embedding_text = ?, updated_at = ?, last_modified = ? WHERE id = ?`) + .bind(dim, now, text, now, now, id).run(); +} +export async function setPersonaNotes(env, id, notes) { + const now = new Date().toISOString(); + await env.DB.prepare(`UPDATE personas SET persona_notes = ?, notes_generated_at = ?, updated_at = ? WHERE id = ?`) + .bind(notes, now, now, id).run(); +} +export async function softDeletePersona(env, id, by) { + const now = new Date().toISOString(); + const r = await env.DB.prepare(`UPDATE personas SET deleted_at = ?, status = 'archived', updated_at = ? WHERE id = ? AND deleted_at IS NULL`).bind(now, now, id).run(); + if ((r.meta?.changes ?? 0) > 0) { + await env.DB.prepare(`INSERT INTO persona_history (id, persona_id, field, new_value, changed_by) VALUES (?, ?, 'archived', ?, ?)`) + .bind(crypto.randomUUID(), id, now, by ?? null).run(); + return true; + } + return false; +} +export async function upsertMatch(env, args) { + const now = new Date().toISOString(); + await env.DB.prepare(`INSERT INTO persona_matches (persona_id, entity_kind, entity_id, fit_score, hard_filter_pass, components_json, explanation, explanation_at, persona_modified_at, entity_modified_at, computed_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(persona_id, entity_kind, entity_id) DO UPDATE SET + fit_score = excluded.fit_score, + hard_filter_pass = excluded.hard_filter_pass, + components_json = excluded.components_json, + -- Drop the cached AI explanation when the new score falls below + -- the explanation threshold so we don't keep stale "why this + -- fits" rationale next to a low score. When the new score is + -- still high but no fresh explanation was generated this pass + -- (e.g. budget cap), keep the previous text — explanation_at + -- preserves the original timestamp so callers can detect age. + explanation = CASE + WHEN excluded.fit_score < 50 THEN NULL + WHEN excluded.explanation IS NOT NULL THEN excluded.explanation + ELSE persona_matches.explanation + END, + explanation_at = CASE + WHEN excluded.fit_score < 50 THEN NULL + WHEN excluded.explanation IS NOT NULL THEN excluded.explanation_at + ELSE persona_matches.explanation_at + END, + persona_modified_at = excluded.persona_modified_at, + entity_modified_at = excluded.entity_modified_at, + computed_at = excluded.computed_at`).bind(args.persona_id, args.entity_kind, args.entity_id, args.fit_score, args.hard_filter_pass, JSON.stringify(args.components), args.explanation, args.explanation ? now : null, args.persona_modified_at, args.entity_modified_at, now).run(); +} +export async function listMatches(env, personaId, opts) { + const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); + const offset = Math.max(0, opts.offset ?? 0); + const minScore = Math.max(0, opts.minScore ?? 0); + const kind = opts.kind ?? "account"; + if (kind === "account") { + const r = await env.DB.prepare(`SELECT pm.*, a.name AS entity_name, a.domain AS entity_domain, a.industry AS entity_industry, a.employees AS entity_employees, a.account_score AS entity_account_score + FROM persona_matches pm + JOIN accounts a ON a.id = pm.entity_id + WHERE pm.persona_id = ? AND pm.entity_kind = 'account' AND pm.fit_score >= ? + ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return r.results ?? []; + } + const r = await env.DB.prepare(`SELECT pm.*, b.name AS entity_name, b.title AS entity_title, b.seniority AS entity_seniority, b.account_id AS entity_account_id + FROM persona_matches pm + JOIN buyers b ON b.id = pm.entity_id + WHERE pm.persona_id = ? AND pm.entity_kind = 'buyer' AND pm.fit_score >= ? + ORDER BY pm.fit_score DESC LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return r.results ?? []; +} +export async function countMatches(env, personaId, minScore = 60) { + const r = await env.DB.prepare(`SELECT COUNT(*) AS c FROM persona_matches WHERE persona_id = ? AND fit_score >= ?`).bind(personaId, minScore).first(); + return r?.c ?? 0; +} +export async function deleteMatchesForPersona(env, personaId) { + await env.DB.prepare(`DELETE FROM persona_matches WHERE persona_id = ?`).bind(personaId).run(); +} +export async function listMatchesForEntity(env, entityKind, entityId) { + const r = await env.DB.prepare(`SELECT persona_id, fit_score FROM persona_matches + WHERE entity_kind = ? AND entity_id = ? + ORDER BY fit_score DESC`).bind(entityKind, entityId).all(); + return r.results ?? []; +} +// Task #58: surface persona-fit on the account/buyer detail pages. +// Joins persona_matches with personas so the dashboard can render a +// "Persona fit" panel without a second round-trip per row. Returns +// rows above `minScore` (default 50, matching the explanation cache +// floor in upsertMatch) sorted by score desc. Skips archived/deleted +// personas — matches against an archived persona are stale evidence. +export async function listMatchesForEntityWithDetails(env, entityKind, entityId, opts) { + const minScore = Math.max(0, opts?.minScore ?? 50); + const personaKind = opts?.personaKind ?? entityKind; + const r = await env.DB.prepare(`SELECT pm.persona_id, pm.fit_score, pm.hard_filter_pass, pm.components_json, + pm.explanation, pm.explanation_at, pm.computed_at, + p.name AS persona_name, p.kind AS persona_kind, + p.status AS persona_status, p.thesis AS persona_thesis + FROM persona_matches pm + JOIN personas p ON p.id = pm.persona_id + WHERE pm.entity_kind = ? AND pm.entity_id = ? + AND pm.fit_score >= ? + AND p.deleted_at IS NULL + AND p.status = 'active' + AND p.kind = ? + ORDER BY pm.fit_score DESC`).bind(entityKind, entityId, minScore, personaKind).all(); + return (r.results ?? []).map((row) => ({ + persona_id: row.persona_id, + persona_name: row.persona_name, + persona_kind: row.persona_kind, + persona_status: row.persona_status, + persona_thesis: row.persona_thesis, + fit_score: row.fit_score, + hard_filter_pass: row.hard_filter_pass, + components: row.components_json ? (() => { try { + return JSON.parse(row.components_json); + } + catch { + return null; + } })() : null, + explanation: row.explanation, + explanation_at: row.explanation_at, + computed_at: row.computed_at, + })); +} +// ----- entity fact loaders shared by scorer + workflow +export async function loadAccountFacts(env, accountId) { + const a = await env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id = ?`).bind(accountId).first(); + if (!a) + return null; + const tech = await env.DB.prepare(`SELECT vendor FROM account_tech WHERE account_id = ?`).bind(accountId).all(); + const sigs = await env.DB.prepare(`SELECT kind, weight, confidence, occurred_at FROM signals WHERE account_id = ? ORDER BY occurred_at DESC LIMIT 200`).bind(accountId).all(); + const buyers = await env.DB.prepare(`SELECT id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id = ? LIMIT 50`).bind(accountId).all(); + const facts = { + status: a.status, + domain: a.domain, + hq_country_iso2: a.hq_country_iso2, + size_band: a.size_band, + employees: a.employees, + industry: a.industry, + industries: parseJsonArr(a.industries_json), + funding_stage: a.funding_stage, + techs: (tech.results ?? []).map((t) => t.vendor), + signals: sigs.results ?? [], + buyers: (buyers.results ?? []).map((b) => ({ + account: null, title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + })), + last_modified: a.updated_at, + }; + return { name: a.name, facts, last_modified: a.updated_at }; +} +export async function loadBuyerFacts(env, buyerId) { + const b = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id = ?`).bind(buyerId).first(); + if (!b) + return null; + const acct = await loadAccountFacts(env, b.account_id); + const facts = { + account: acct?.facts ?? null, + title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + }; + return { name: b.name ?? b.title ?? buyerId, facts, last_modified: b.updated_at, account_id: b.account_id }; +} +// Bulk loader for AccountFacts. Issues 4 set-based queries (accounts + +// account_tech + signals + buyers, each with WHERE id IN (?...)) and +// stitches them together. Used by rescorePersonaFull and the +// /preview endpoint to avoid N round-trips over the D1 binding. +export async function loadAccountFactsBulk(env, ids) { + const out = new Map(); + if (!ids.length) + return out; + const ph = ids.map(() => "?").join(","); + const [a, tech, sigs, buyers] = await Promise.all([ + env.DB.prepare(`SELECT id, name, status, domain, hq_country_iso2, size_band, employees, industry, industries_json, funding_stage, updated_at FROM accounts WHERE id IN (${ph})`).bind(...ids).all(), + env.DB.prepare(`SELECT account_id, vendor FROM account_tech WHERE account_id IN (${ph})`).bind(...ids).all(), + env.DB.prepare(`SELECT account_id, kind, weight, confidence, occurred_at FROM signals WHERE account_id IN (${ph}) ORDER BY occurred_at DESC`).bind(...ids).all(), + env.DB.prepare(`SELECT id, account_id, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE account_id IN (${ph})`).bind(...ids).all(), + ]); + const techByAcct = new Map(); + for (const t of tech.results ?? []) { + const arr = techByAcct.get(t.account_id) ?? []; + arr.push(t.vendor); + techByAcct.set(t.account_id, arr); + } + const sigByAcct = new Map(); + for (const s of sigs.results ?? []) { + const arr = sigByAcct.get(s.account_id) ?? []; + if (arr.length < 200) + arr.push({ kind: s.kind, weight: s.weight, confidence: s.confidence, occurred_at: s.occurred_at }); + sigByAcct.set(s.account_id, arr); + } + const buyByAcct = new Map(); + for (const b of buyers.results ?? []) { + const arr = buyByAcct.get(b.account_id) ?? []; + if (arr.length < 50) + arr.push({ account: null, title: b.title, seniority: b.seniority, department: b.department, is_decision_maker: b.is_decision_maker, last_modified: b.updated_at }); + buyByAcct.set(b.account_id, arr); + } + for (const row of a.results ?? []) { + const facts = { + status: row.status, domain: row.domain, hq_country_iso2: row.hq_country_iso2, + size_band: row.size_band, employees: row.employees, industry: row.industry, + industries: parseJsonArr(row.industries_json), funding_stage: row.funding_stage, + techs: techByAcct.get(row.id) ?? [], + signals: sigByAcct.get(row.id) ?? [], + buyers: buyByAcct.get(row.id) ?? [], + last_modified: row.updated_at, + }; + out.set(row.id, { name: row.name, facts, last_modified: row.updated_at }); + } + return out; +} +// Bulk loader for BuyerFacts. Issues 1 query for the buyers + delegates +// to loadAccountFactsBulk for parent accounts (one round-trip via IN). +export async function loadBuyerFactsBulk(env, ids) { + const out = new Map(); + if (!ids.length) + return out; + const ph = ids.map(() => "?").join(","); + const r = await env.DB.prepare(`SELECT id, account_id, name, title, seniority, department, is_decision_maker, updated_at FROM buyers WHERE id IN (${ph})`).bind(...ids).all(); + const buyerRows = r.results ?? []; + const acctIds = Array.from(new Set(buyerRows.map((b) => b.account_id))); + const acctFacts = await loadAccountFactsBulk(env, acctIds); + for (const b of buyerRows) { + const acct = acctFacts.get(b.account_id); + const facts = { + account: acct?.facts ?? null, title: b.title, seniority: b.seniority, department: b.department, + is_decision_maker: b.is_decision_maker, last_modified: b.updated_at, + }; + out.set(b.id, { name: b.name ?? b.title ?? b.id, facts, last_modified: b.updated_at, account_id: b.account_id }); + } + return out; +} +// Bulk writeback: recompute max active-persona fit_score for each id +// in one aggregate query, then UPDATE in one statement per kind. Used +// at the end of each rescore batch instead of N per-row writebacks. +export async function bulkWriteBackFit(env, kind, ids) { + if (!ids.length) + return; + const ph = ids.map(() => "?").join(","); + const rows = await env.DB.prepare(`SELECT pm.entity_id AS id, MAX(pm.fit_score) AS m + FROM persona_matches pm + JOIN personas p ON p.id = pm.persona_id + WHERE pm.entity_kind = ? AND pm.entity_id IN (${ph}) + AND p.status = 'active' AND p.deleted_at IS NULL + GROUP BY pm.entity_id`).bind(kind, ...ids).all(); + const maxById = new Map(); + for (const r of rows.results ?? []) + maxById.set(r.id, r.m ?? 0); + // Issue updates as a batch (each binds its own params; D1 batches + // these into one HTTP round-trip via the binding's batch() API). + const stmts = ids.map((id) => { + const m = maxById.get(id) ?? 0; + if (kind === "account") { + return env.DB.prepare(`UPDATE accounts SET fit_score = ?, account_score = ROUND((0.6 * intent_score) + (0.4 * ?), 2) WHERE id = ?`).bind(m, m, id); + } + return env.DB.prepare(`UPDATE buyers SET fit_score = ? WHERE id = ?`).bind(m, id); + }); + await env.DB.batch(stmts); +} +// Materialize a compact "facts" object for the AI explainer. +export function summarizeAccountForExplanation(name, f) { + return { + name, + domain: f.domain, + industry: f.industry, + employees: f.employees, + size_band: f.size_band, + country: f.hq_country_iso2, + funding_stage: f.funding_stage, + top_techs: f.techs.slice(0, 8), + top_signals: f.signals.slice(0, 5).map((s) => ({ kind: s.kind, weight: s.weight, occurred_at: s.occurred_at })), + top_buyer: f.buyers[0] ? { title: f.buyers[0].title, seniority: f.buyers[0].seniority } : null, + }; +} +export function summarizeBuyerForExplanation(name, f) { + return { + name, + title: f.title, + seniority: f.seniority, + department: f.department, + is_decision_maker: !!f.is_decision_maker, + account: f.account ? summarizeAccountForExplanation(name, f.account) : null, + }; +} diff --git a/apps/worker/test-dist-q/personas/score.js b/apps/worker/test-dist-q/personas/score.js new file mode 100644 index 00000000..4cd5e110 --- /dev/null +++ b/apps/worker/test-dist-q/personas/score.js @@ -0,0 +1,281 @@ +// Task #46: deterministic persona scoring. +// +// fit_score = clamp(0..100, recency_boost * Σ w_i * c_i) when hard +// filters pass, else 0. Components are each 0..100. semantic_fit comes +// from cosine similarity against the entity's existing embedding (we +// pass it in pre-computed); the rest are computed here from row data. +export const DEFAULT_WEIGHTS_ACCOUNT = { + size: 0.10, + geo: 0.10, + industry: 0.20, + tech: 0.10, + signal: 0.20, + buyer: 0.10, + semantic: 0.20, +}; +export const DEFAULT_WEIGHTS_BUYER = { + size: 0.05, + geo: 0.05, + industry: 0.15, + tech: 0.05, + signal: 0.10, + buyer: 0.45, + semantic: 0.15, +}; +const DAY = 86_400_000; +function lc(s) { return (s ?? "").toLowerCase().trim(); } +function clamp(n, lo, hi) { return Math.max(lo, Math.min(hi, n)); } +function scoreSize(p, emp, band) { + if (p.size_min == null && p.size_max == null && p.size_bands.length === 0) + return 50; + if (band && p.size_bands.length && p.size_bands.includes(band)) + return 100; + if (emp == null) + return 25; + const lo = p.size_min ?? 0; + const hi = p.size_max ?? Number.MAX_SAFE_INTEGER; + if (emp >= lo && emp <= hi) + return 100; + // Soft penalty: 25% per order-of-magnitude away. + const ratio = emp < lo ? lo / Math.max(1, emp) : emp / hi; + const decades = Math.log10(ratio); + return Math.max(0, 100 - Math.round(decades * 60)); +} +function scoreGeo(p, iso) { + if (!p.geos.length) + return 50; + if (!iso) + return 20; + const i = lc(iso); + if (p.geos.includes(i)) + return 100; + // Region bucketing + const REGION = { + emea: ["gb", "ie", "fr", "de", "es", "it", "nl", "be", "se", "no", "fi", "dk", "pt", "pl", "ch", "at"], + apac: ["jp", "sg", "au", "nz", "kr", "in", "hk", "tw", "my", "id", "ph", "th", "vn"], + latam: ["br", "mx", "ar", "cl", "co", "pe", "uy"], + africa: ["za", "ng", "ke", "eg", "ma"], + }; + for (const slug of p.geos) { + const ctry = REGION[slug]; + if (ctry && ctry.includes(i)) + return 80; + if (slug === "global") + return 60; + } + return 0; +} +function scoreIndustry(p, primary, all) { + if (!p.industries.length) + return 50; + const want = new Set(p.industries.map(lc)); + if (primary && want.has(lc(primary))) + return 100; + for (const i of all.map(lc)) + if (want.has(i)) + return 80; + return 0; +} +function scoreTech(p, techs) { + const have = new Set(techs.map(lc)); + if (p.techs_excluded.some((t) => have.has(lc(t)))) + return { score: 0, pass: false }; + if (p.techs_required.length) { + const all = p.techs_required.every((t) => have.has(lc(t))); + if (!all) + return { score: 0, pass: false }; + } + if (!p.techs_preferred.length && !p.techs_required.length) + return { score: 50, pass: true }; + const matched = p.techs_preferred.filter((t) => have.has(lc(t))).length; + const ratio = p.techs_preferred.length ? matched / p.techs_preferred.length : 1; + return { score: Math.round(60 + 40 * ratio), pass: true }; +} +function scoreSignals(p, sigs) { + if (!sigs.length) + return 0; + const want = new Set(p.signal_kinds.map(lc)); + const now = Date.now(); + let totalCredit = 0; + for (const s of sigs) { + const t = Date.parse(s.occurred_at); + const age = Number.isFinite(t) ? Math.max(0, (now - t) / DAY) : 365; + const decay = Math.exp(-age / 30); // half-life-ish + const w = (s.weight ?? 0) * (s.confidence ?? 1); + const kindMul = want.size === 0 ? 0.5 : want.has(lc(s.kind)) ? 1.0 : 0.25; + totalCredit += w * decay * kindMul; + } + return Math.round(100 * (1 - Math.exp(-totalCredit / 15))); +} +function scoreBuyer(p, b) { + let s = 0; + let denom = 0; + if (p.buyer_titles.length) { + denom += 50; + const t = lc(b.title); + if (t && p.buyer_titles.some((x) => t.includes(lc(x)))) + s += 50; + } + if (p.buyer_seniority.length) { + denom += 30; + if (b.seniority && p.buyer_seniority.includes(lc(b.seniority))) + s += 30; + } + if (p.buyer_departments.length) { + denom += 20; + if (b.department && p.buyer_departments.includes(lc(b.department))) + s += 20; + } + if (denom === 0) + return 50; + // Decision-maker bonus + if (b.is_decision_maker) + s = Math.min(denom, s + 5); + return Math.round((s / denom) * 100); +} +function scoreBuyersForAccount(p, buyers) { + if (!buyers.length) + return p.buyer_titles.length || p.buyer_seniority.length ? 0 : 50; + let best = 0; + for (const b of buyers) + best = Math.max(best, scoreBuyer(p, b)); + return best; +} +export function checkHardFilters(p, account, buyer) { + const reasons = []; + const f = p.hard_filters || {}; + const acc = account ?? buyer?.account ?? null; + if (f.require_domain && acc && !acc.domain) { + reasons.push("missing_domain"); + return { pass: false, reasons }; + } + if (Array.isArray(f.statuses_in) && acc && !f.statuses_in.map(lc).includes(lc(acc.status))) { + reasons.push(`status_not_in:${acc.status}`); + return { pass: false, reasons }; + } + if (Array.isArray(f.exclude_country_iso2) && acc && acc.hq_country_iso2 && f.exclude_country_iso2.map(lc).includes(lc(acc.hq_country_iso2))) { + reasons.push(`country_excluded:${acc.hq_country_iso2}`); + return { pass: false, reasons }; + } + if (Array.isArray(f.country_iso2_in) && acc && (!acc.hq_country_iso2 || !f.country_iso2_in.map(lc).includes(lc(acc.hq_country_iso2)))) { + reasons.push("country_not_in"); + return { pass: false, reasons }; + } + if (Array.isArray(f.funding_stage_in) && acc && (!acc.funding_stage || !f.funding_stage_in.map(lc).includes(lc(acc.funding_stage)))) { + reasons.push("funding_stage_not_in"); + return { pass: false, reasons }; + } + if (f.is_decision_maker && buyer && !buyer.is_decision_maker) { + reasons.push("not_decision_maker"); + return { pass: false, reasons }; + } + return { pass: true, reasons }; +} +export function recencyBoost(p, lastModifiedISO) { + // Override wins; otherwise small boost (max 1.15x) for entities updated + // within the last 7 days. Capped 1.0..1.2 per spec. + if (p.recency_boost && p.recency_boost > 0) + return clamp(p.recency_boost, 1.0, 1.2); + if (!lastModifiedISO) + return 1.0; + const t = Date.parse(lastModifiedISO); + if (!Number.isFinite(t)) + return 1.0; + const ageDays = Math.max(0, (Date.now() - t) / DAY); + if (ageDays <= 7) + return 1.15; + if (ageDays <= 30) + return 1.05; + return 1.0; +} +export function scoreEntity(p, ctx) { + const reasons = []; + const hf = checkHardFilters(p, ctx.account, ctx.buyer); + if (!hf.pass) { + return { + fit_score: 0, + components: { + hard_filter_pass: 0, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, + signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, + weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, + reasons: hf.reasons, + }, + }; + } + const acc = ctx.account ?? ctx.buyer?.account ?? null; + const tech = scoreTech(p, acc?.techs ?? []); + if (!tech.pass) { + reasons.push("tech_excluded_or_missing_required"); + return { + fit_score: 0, + components: { + hard_filter_pass: 1, size_fit: 0, geo_fit: 0, industry_fit: 0, tech_fit: 0, + signal_fit: 0, buyer_fit: 0, semantic_fit: 0, recency_boost: 1, + weights: { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }, + reasons, + }, + }; + } + const size = scoreSize(p, acc?.employees ?? null, acc?.size_band ?? null); + const geo = scoreGeo(p, acc?.hq_country_iso2 ?? null); + const industry = scoreIndustry(p, acc?.industry ?? null, acc?.industries ?? []); + const signal = scoreSignals(p, acc?.signals ?? []); + const buyerScore = ctx.buyer ? scoreBuyer(p, ctx.buyer) : scoreBuyersForAccount(p, acc?.buyers ?? []); + const semCos = typeof ctx.semanticCosine === "number" ? ctx.semanticCosine : null; + const semantic = semCos == null + ? 50 + : semCos < (p.semantic_fit_threshold ?? 0.55) ? 0 : Math.round(clamp((semCos - 0.4) / 0.5, 0, 1) * 100); + const w = { ...(p.kind === "account" ? DEFAULT_WEIGHTS_ACCOUNT : DEFAULT_WEIGHTS_BUYER), ...p.weights }; + const sumW = (w.size ?? 0) + (w.geo ?? 0) + (w.industry ?? 0) + (w.tech ?? 0) + (w.signal ?? 0) + (w.buyer ?? 0) + (w.semantic ?? 0); + const norm = sumW > 0 ? sumW : 1; + const blended = ((size * (w.size ?? 0)) + + (geo * (w.geo ?? 0)) + + (industry * (w.industry ?? 0)) + + (tech.score * (w.tech ?? 0)) + + (signal * (w.signal ?? 0)) + + (buyerScore * (w.buyer ?? 0)) + + (semantic * (w.semantic ?? 0))) / norm; + const boost = recencyBoost(p, ctx.buyer?.last_modified ?? acc?.last_modified ?? null); + const fit = Math.round(clamp(blended * boost, 0, 100)); + return { + fit_score: fit, + components: { + hard_filter_pass: 1, + size_fit: size, + geo_fit: geo, + industry_fit: industry, + tech_fit: tech.score, + signal_fit: signal, + buyer_fit: buyerScore, + semantic_fit: semantic, + recency_boost: boost, + weights: w, + reasons, + }, + }; +} +export function buildEmbeddingText(p) { + const parts = []; + parts.push(`Persona: ${p.name}`); + if (p.thesis) + parts.push(`Thesis: ${p.thesis}`); + if (p.industries.length) + parts.push(`Industries: ${p.industries.join(", ")}`); + if (p.geos.length) + parts.push(`Geos: ${p.geos.join(", ")}`); + if (p.size_min || p.size_max) + parts.push(`Size: ${p.size_min ?? "?"}–${p.size_max ?? "?"} FTE`); + if (p.techs_required.length) + parts.push(`Required tech: ${p.techs_required.join(", ")}`); + if (p.techs_preferred.length) + parts.push(`Preferred tech: ${p.techs_preferred.join(", ")}`); + if (p.signal_kinds.length) + parts.push(`Watch signals: ${p.signal_kinds.join(", ")}`); + if (p.buyer_titles.length) + parts.push(`Buyer titles: ${p.buyer_titles.join(", ")}`); + if (p.buyer_seniority.length) + parts.push(`Buyer seniority: ${p.buyer_seniority.join(", ")}`); + if (p.buyer_departments.length) + parts.push(`Buyer departments: ${p.buyer_departments.join(", ")}`); + return parts.join(" | "); +} diff --git a/apps/worker/test-dist-q/scraper/firms_upsert.js b/apps/worker/test-dist-q/scraper/firms_upsert.js new file mode 100644 index 00000000..5dbae1ad --- /dev/null +++ b/apps/worker/test-dist-q/scraper/firms_upsert.js @@ -0,0 +1,267 @@ +import { extractDomain } from "./normalize"; +import { syncFirmToEntity } from "../entities/dualwrite"; +const SCALAR_FIELDS = [ + "legal_name", "kind", "website", "logo_url", + "hq_country_iso2", "hq_region", "hq_city", + "thesis", "check_size_min_usd", "check_size_max_usd", "check_size_typical_usd", + "aum_usd", "fund_count", "current_fund_name", "current_fund_size_usd", + "lead_or_co", "portfolio_count", "founded_year", "team_size", + "linkedin_url", "crunchbase_url", "twitter_handle", + "signal_nfx_url", "openvc_url", "pitchbook_url", + "contact_email", "submission_url", +]; +// `source_url` is intentionally excluded from SCALAR_FIELDS — Task #1 +// requires that re-imports from different Folk shares union the +// provenance URLs at the firm row level rather than fill-if-empty. The +// merge path below comma-joins distinct values (mirrors `imported_from`). +const ARRAY_FIELDS = [ + { key: "geo_focus", column: "geo_focus_json" }, + { key: "stages", column: "stages_json" }, + { key: "sectors", column: "sectors_json" }, + { key: "notable_investments", column: "notable_investments_json" }, +]; +export async function upsertFirm(env, candidate, importedFrom, +/** + * Task #1: optional dual-write provenance override. Folk-share imports + * pass `{ source: 'folk_share', sourceKind: 'import' }` so the firm's + * unified-graph facts carry import provenance instead of the default + * `source_kind='scrape'`. + */ +importCtx) { + const name = candidate.name?.trim(); + if (!name) + throw new Error("upsertFirm: candidate.name required"); + const rawDomain = candidate.domain ?? deriveDomain(candidate.website); + const domain = rawDomain ? rawDomain.toLowerCase().trim() : null; + if (domain) + candidate.domain = domain; + // Quality gate: require name + (domain OR website). Without either, + // dedupe is impossible and reruns would create endless duplicates. + if (!domain && !candidate.website) { + throw new Error("upsertFirm: candidate must have domain or website"); + } + const lname = name.toLowerCase(); + // Dedupe lookup. Match on the effective domain (stored.domain coalesced + // with the parsed hostname of stored.website) so a rerun that supplies + // a domain still merges with a row that originally only had a website. + const candidates = await env.DB.prepare("SELECT * FROM firms WHERE lower(name) = ? LIMIT 50").bind(lname).all(); + const rows = candidates.results ?? []; + let existing = null; + for (const r of rows) { + const storedDomain = r.domain ?? deriveDomain(r.website ?? undefined); + if (domain && storedDomain && storedDomain.toLowerCase() === domain) { + existing = r; + break; + } + if (!domain && !storedDomain) { + existing = r; + break; + } + } + const result = existing + ? await mergeInto(env, existing, candidate, importedFrom) + : await insertNew(env, candidate, domain, importedFrom); + // Task #4: dual-write into the unified entity graph (best-effort — + // never block the legacy firm-list importer on a unified-model error). + try { + await syncFirmToEntity(env, { + id: result.firmId, + name: candidate.name, + legal_name: candidate.legal_name ?? null, + website: result.website, + domain: result.domain, + hq_country_iso2: candidate.hq_country_iso2 ?? null, + hq_region: candidate.hq_region ?? null, + hq_city: candidate.hq_city ?? null, + check_size_min_usd: candidate.check_size_min_usd ?? null, + check_size_max_usd: candidate.check_size_max_usd ?? null, + check_size_typical_usd: candidate.check_size_typical_usd ?? null, + thesis: candidate.thesis ?? null, + linkedin_url: candidate.linkedin_url ?? null, + crunchbase_url: candidate.crunchbase_url ?? null, + twitter_handle: candidate.twitter_handle ?? null, + contact_email: candidate.contact_email ?? null, + sectors_json: candidate.sectors ? JSON.stringify(candidate.sectors) : null, + stages_json: candidate.stages ? JSON.stringify(candidate.stages) : null, + geo_focus_json: candidate.geo_focus ? JSON.stringify(candidate.geo_focus) : null, + kind: candidate.kind ?? null, + }, importCtx?.source ?? importedFrom, importCtx?.sourceKind ?? "scrape"); + } + catch (e) { + console.warn("dualwrite syncFirmToEntity failed", result.firmId, e.message); + } + // Role inference now runs centrally inside syncFirmToEntity (the + // unified entity write path), so we no longer call it here. + return result; +} +async function insertNew(env, c, domain, importedFrom) { + const slug = await pickUniqueSlug(env, c.name, domain); + const cols = ["name", "slug", "domain", "imported_from", "last_modified"]; + const vals = [c.name.trim(), slug, domain, importedFrom, new Date().toISOString()]; + for (const f of SCALAR_FIELDS) { + const v = c[f]; + if (v != null && v !== "") { + cols.push(f); + vals.push(v); + } + } + for (const { key, column } of ARRAY_FIELDS) { + const v = c[key]; + if (v && v.length) { + cols.push(column); + vals.push(JSON.stringify(uniqStringArray(v))); + } + } + if (c.source_url) { + cols.push("source_url"); + vals.push(c.source_url); + } + if (c.socials) { + cols.push("socials_json"); + vals.push(JSON.stringify(c.socials)); + } + if (c.notes) { + cols.push("notes"); + vals.push(c.notes); + } + const placeholders = cols.map(() => "?").join(","); + const r = await env.DB.prepare(`INSERT INTO firms (${cols.join(",")}) VALUES (${placeholders})`).bind(...vals).run(); + const firmId = Number(r.meta.last_row_id); + return { firmId, action: "created", website: c.website ?? null, domain }; +} +async function mergeInto(env, existing, c, importedFrom) { + const sets = []; + const binds = []; + // Scalar: only fill missing values; existing non-null values win. + for (const f of SCALAR_FIELDS) { + const newVal = c[f]; + if (newVal == null || newVal === "") + continue; + if (existing[f] == null || existing[f] === "") { + sets.push(`${f} = ?`); + binds.push(newVal); + } + } + // Array fields: set-union of existing JSON array + new entries. + for (const { key, column } of ARRAY_FIELDS) { + const incoming = c[key]; + if (!incoming || !incoming.length) + continue; + const existingArr = parseJsonArray(existing[column]); + const merged = uniqStringArray([...existingArr, ...incoming]); + if (merged.length !== existingArr.length) { + sets.push(`${column} = ?`); + binds.push(JSON.stringify(merged)); + } + } + if (c.socials) { + const existingSocials = parseJsonObject(existing.socials_json); + const merged = { ...existingSocials, ...c.socials }; + if (Object.keys(merged).length !== Object.keys(existingSocials).length) { + sets.push("socials_json = ?"); + binds.push(JSON.stringify(merged)); + } + } + if (c.notes && !existing.notes) { + sets.push("notes = ?"); + binds.push(c.notes); + } + // Track every distinct origin. + const importedFromExisting = existing.imported_from ?? ""; + if (!importedFromExisting.split(",").includes(importedFrom)) { + sets.push("imported_from = ?"); + binds.push(importedFromExisting ? `${importedFromExisting},${importedFrom}` : importedFrom); + } + // Task #1: union source_url across re-imports. Folk shares (Top-300, + // FR VCs, etc.) each have their own share URL; re-importing the same + // firm from a second share must preserve evidence of both shares + // rather than fill-if-empty (which would silently drop the second + // URL). Mirrors the imported_from comma-join pattern above; the + // unified graph still gets one channel/fact per share via dualwrite. + const newSourceUrl = c.source_url; + if (typeof newSourceUrl === "string" && newSourceUrl) { + const existingSourceUrl = existing.source_url ?? ""; + const parts = existingSourceUrl ? existingSourceUrl.split(",").map((s) => s.trim()).filter(Boolean) : []; + if (!parts.includes(newSourceUrl)) { + parts.push(newSourceUrl); + sets.push("source_url = ?"); + binds.push(parts.join(",")); + } + } + // Always bump last_modified on every dedupe hit — even when no field + // deltas applied — so reruns leave a verifiable timestamp trail. + const action = sets.length ? "updated" : "unchanged"; + sets.push("last_modified = ?"); + binds.push(new Date().toISOString()); + binds.push(existing.id); + await env.DB.prepare(`UPDATE firms SET ${sets.join(", ")} WHERE id = ?`).bind(...binds).run(); + const persistedWebsite = existing.website ?? c.website ?? null; + const persistedDomain = existing.domain ?? deriveDomain(persistedWebsite); + return { firmId: existing.id, action, website: persistedWebsite, domain: persistedDomain }; +} +function deriveDomain(website) { + if (!website) + return null; + return extractDomain(website) || null; +} +function slugify(s) { + return s.toLowerCase().normalize("NFKD").replace(/[^\w]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80); +} +async function pickUniqueSlug(env, name, domain) { + const base = slugify(name) || "firm"; + let candidate = base; + let row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); + if (!row) + return candidate; + if (domain) { + candidate = `${base}-${slugify(domain)}`; + row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(candidate).first(); + if (!row) + return candidate; + } + for (let i = 2; i < 100; i++) { + const c = `${base}-${i}`; + row = await env.DB.prepare("SELECT 1 AS x FROM firms WHERE slug = ? LIMIT 1").bind(c).first(); + if (!row) + return c; + } + // Last resort: random suffix. + return `${base}-${Math.random().toString(36).slice(2, 8)}`; +} +function parseJsonArray(raw) { + if (!raw) + return []; + try { + const v = JSON.parse(raw); + return Array.isArray(v) ? v.map((x) => String(x)) : []; + } + catch { + return []; + } +} +function parseJsonObject(raw) { + if (!raw) + return {}; + try { + const v = JSON.parse(raw); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +function uniqStringArray(arr) { + const seen = new Set(); + const out = []; + for (const s of arr) { + const k = String(s).trim(); + if (!k) + continue; + const lk = k.toLowerCase(); + if (seen.has(lk)) + continue; + seen.add(lk); + out.push(k); + } + return out; +} diff --git a/apps/worker/test-dist-q/scraper/normalize.js b/apps/worker/test-dist-q/scraper/normalize.js new file mode 100644 index 00000000..5f9c29b0 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/normalize.js @@ -0,0 +1,104 @@ +// Normalization helpers used by the parsers and dedupe key generators. +const EMAIL_RE = /^[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}$/i; +const PHONE_DIGITS_RE = /[^\d+]/g; +export function normalizeEmail(raw) { + if (!raw) + return null; + const s = raw.trim().toLowerCase(); + if (!EMAIL_RE.test(s)) + return null; + return s; +} +/** + * Email key for dedupe purposes only — strips +tags and lowercases the domain. + * The displayed email uses normalizeEmail (preserves the +tag). + */ +export function emailDedupeKey(email) { + const e = normalizeEmail(email); + if (!e) + return null; + const [localPart, domain] = e.split("@"); + const stripped = localPart.split("+")[0]; + return `${stripped}@${domain.toLowerCase()}`; +} +/** + * Best-effort E.164 normalization. We only accept numbers that already include + * a leading + or are clearly internationalizable (10–15 digits). Otherwise null. + */ +export function normalizePhoneE164(raw) { + if (!raw) + return null; + const cleaned = raw.replace(PHONE_DIGITS_RE, ""); + if (!cleaned) + return null; + if (cleaned.startsWith("+")) { + const digits = cleaned.slice(1); + if (digits.length < 7 || digits.length > 15) + return null; + return `+${digits}`; + } + // No country code → don't guess; return null so we don't pollute dedupe keys. + if (cleaned.length === 10 || cleaned.length === 11) { + // Common case: US/Canada 10 digits or 1+10 digits. + const d = cleaned.length === 10 ? `1${cleaned}` : cleaned; + return `+${d}`; + } + return null; +} +export function canonicalizeLinkedinUrl(url) { + if (!url) + return null; + try { + const u = new URL(url); + const host = u.hostname.toLowerCase().replace(/^www\./, ""); + if (!host.endsWith("linkedin.com")) + return null; + // Trailing slash and query strip + const path = u.pathname.replace(/\/+$/, "").toLowerCase(); + if (!/^\/(in|company|school)\/[^\/]+/.test(path)) + return null; + return `https://www.linkedin.com${path}`; + } + catch { + return null; + } +} +export function countryNameToIso2(name) { + if (!name) + return null; + const s = name.trim().toLowerCase(); + // Tiny seed map; the full mapping lives with the taxonomies task. + const seed = { + usa: "US", + "united states": "US", + "united states of america": "US", + canada: "CA", + uk: "GB", + "united kingdom": "GB", + england: "GB", + france: "FR", + germany: "DE", + spain: "ES", + italy: "IT", + netherlands: "NL", + switzerland: "CH", + israel: "IL", + india: "IN", + china: "CN", + japan: "JP", + singapore: "SG", + australia: "AU", + brazil: "BR", + }; + if (s.length === 2) + return s.toUpperCase(); + return seed[s] ?? null; +} +export function extractDomain(url) { + try { + return new URL(url).hostname.toLowerCase().replace(/^www\./, ""); + } + catch { + return ""; + } +} diff --git a/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js b/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js new file mode 100644 index 00000000..cb0ff5c3 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/parsers/firmlists/types.js @@ -0,0 +1 @@ +export {}; diff --git a/apps/worker/test-dist-q/scraper/rateLimit.js b/apps/worker/test-dist-q/scraper/rateLimit.js new file mode 100644 index 00000000..ebd48fb8 --- /dev/null +++ b/apps/worker/test-dist-q/scraper/rateLimit.js @@ -0,0 +1,53 @@ +// Per-host + global AI rate limiting (Task #25 step 6). +// +// Prefers the Cloudflare Rate Limiter binding (`RL_HOST`/`RL_AI`) when +// configured. Falls back to a KV-backed leaky-bucket counter on +// SCRAPE_CACHE so the worker keeps pacing itself even when the binding is +// missing in dev or before the namespace is provisioned. +const KV_PREFIX = "rl:"; +const HOST_LIMIT_PER_MIN = 60; +const AI_LIMIT_PER_MIN = 600; +const WINDOW_MS = 60_000; +async function kvLeakyBucket(kv, key, limit) { + if (!kv) + return true; + const raw = await kv.get(key); + const now = Date.now(); + let bucket = raw ? safeParse(raw) : { count: 0, window_start: now }; + if (now - bucket.window_start > WINDOW_MS) + bucket = { count: 0, window_start: now }; + if (bucket.count >= limit) + return false; + bucket.count += 1; + await kv.put(key, JSON.stringify(bucket), { expirationTtl: 120 }); + return true; +} +function safeParse(raw) { + try { + const v = JSON.parse(raw); + if (typeof v?.count === "number" && typeof v?.window_start === "number") + return v; + } + catch { /* swallow */ } + return { count: 0, window_start: Date.now() }; +} +export async function limitHost(env, host) { + if (env.RL_HOST) { + try { + const r = await env.RL_HOST.limit({ key: host }); + return r.success; + } + catch { /* fall through to KV */ } + } + return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}host:${host}`, HOST_LIMIT_PER_MIN); +} +export async function limitAi(env) { + if (env.RL_AI) { + try { + const r = await env.RL_AI.limit({ key: "global" }); + return r.success; + } + catch { /* fall through */ } + } + return kvLeakyBucket(env.SCRAPE_CACHE, `${KV_PREFIX}ai:global`, AI_LIMIT_PER_MIN); +} diff --git a/apps/worker/test-dist-q/services/personaMatchTrigger.js b/apps/worker/test-dist-q/services/personaMatchTrigger.js new file mode 100644 index 00000000..16f48214 --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatchTrigger.js @@ -0,0 +1,59 @@ +// Task #8: per-entity persona-match refresh trigger. +// +// Called from entity write paths (insertFact, addCareerEntry) so a new +// job or relocation flows into persona candidate rankings within +// minutes, not at the next nightly cron. Debounced via KV so a burst +// of fact writes for the same entity only triggers one re-match. +// Predicates that materially affect a person-entity's persona score. +// Other predicates (donations, family ties, lifestyle, etc.) are +// ignored here so we don't dispatch on unrelated edits. +const RELEVANT_PREDICATES = new Set([ + "person.career", "person.title", "title", + "person.seniority", "person.department", + "person.location.country", "person.location.city", + "location.country", "location.city", + "employer", "person.employer", + // `employees` is the spelling that actually gets written; without it a + // fresh headcount fact never re-scored the entity it belongs to. + "employees", + "org.headcount", "org.employees", + "company.employees", "company.headcount", + "org.sector", "sector", + "org.stage", "stage", +]); +export function isRelevantPredicate(predicate) { + if (!predicate) + return false; + return RELEVANT_PREDICATES.has(predicate); +} +const DEBOUNCE_SECONDS = 300; // 5 minutes +export async function triggerEntityMatchRefresh(env, entityId) { + if (!entityId) + return; + // KV debounce — first write wins per 5min window. + try { + const kvKey = `pem:trigger:${entityId}`; + if (env.SESSIONS) { + const existing = await env.SESSIONS.get(kvKey); + if (existing) + return; + await env.SESSIONS.put(kvKey, "1", { expirationTtl: DEBOUNCE_SECONDS }); + } + } + catch (e) { + console.warn("triggerEntityMatchRefresh debounce check failed", entityId, e.message); + // Fall through — better to dispatch than miss the trigger. + } + // Dispatch the per-entity workflow; inline fallback runs the service. + try { + if (env.WF_PERSONA_MATCH_ENTITY) { + await env.WF_PERSONA_MATCH_ENTITY.create({ params: { entityId } }); + return; + } + const { scoreEntityAcrossPersonas } = await import("./personaMatching.js"); + await scoreEntityAcrossPersonas(env, entityId); + } + catch (e) { + console.warn("triggerEntityMatchRefresh dispatch failed", entityId, e.message); + } +} diff --git a/apps/worker/test-dist-q/services/personaMatching.js b/apps/worker/test-dist-q/services/personaMatching.js new file mode 100644 index 00000000..c2e893c1 --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatching.js @@ -0,0 +1,484 @@ +// Task #8: Real persona matching algorithm. +// +// Deterministic weighted scoring engine that ranks unified `u_entities` +// (person entities) against a persona. Each entity gets a score in +// [0,1] plus a transparent per-component breakdown so the dashboard +// can explain *why* an entity matched. +// +// Pure scoring primitives live in personaMatchingScorers.ts (no Env +// imports — unit-testable). This module orchestrates the D1 loads, +// the title embedding, and the upsert. +import { aiEmbed } from "../ai/extract"; +import { assertBudget } from "../ai/budget"; +import { getPersona } from "../personas/repo"; +import { DEFAULT_WEIGHTS, MODEL_VERSION, cosine, aggregate, buildRationale, extractTargets, scoreSeniority, scoreFunction, scoreIndustry, scoreCompanySize, scoreStage, scoreGeo, } from "./personaMatchingScorers"; +export { DEFAULT_WEIGHTS, MODEL_VERSION, extractTargets }; +// Task #3: structural-only fallback used when a kind plugin returns +// null (e.g. fund/company targets that have no per-entity scoring +// pipeline). Builds a properly-typed ComponentMap with zeroed +// components so downstream consumers (rationale builder, persistence) +// don't have to special-case the structural row. Replaces an earlier +// `as unknown as MatchResult` cast that bypassed the type system. +export function buildStructuralFallback(reason) { + const components = {}; + for (const key of Object.keys(DEFAULT_WEIGHTS)) { + components[key] = { value: 0, weight: 0, reason: "n/a (structural fallback)" }; + } + return { score: 0.5, components, rationale: reason }; +} +// --------------------------------------------------------------------------- +// Entity loader. +// --------------------------------------------------------------------------- +async function loadEmployerFacts(env, employerId) { + const sum = await env.DB.prepare(`SELECT display_name, country_iso2, sectors_csv, stages_csv FROM entity_summary WHERE entity_id = ?`).bind(employerId).first(); + // `employees` is first because it is the only one of these that anything + // writes: it is the predicate the registry declares + // (entities/profile-predicates.ts) and the one secEdgar/persist.ts and the + // account dual-write emit. The four `org.*` / `company.*` spellings below + // were the entire list, and no writer has ever produced one — so + // `employees` came back null for every entity and scoreCompanySize + // returned its "company size unknown" zero every time. With a weight of + // 0.10 that put a hard ceiling of 0.90 on every persona match, and made + // "company size unknown" a permanent line in the rationale the dashboard + // shows to explain why someone matched. The unused spellings are kept so a + // future writer picking one still resolves. + const hc = await env.DB.prepare(`SELECT value_number FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('employees','org.headcount','org.employees','company.employees','company.headcount') AND value_number IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(employerId).first(); + if (!sum && !hc) + return null; + return { + name: sum?.display_name ?? null, + country: sum?.country_iso2 ?? null, + sectors: sum?.sectors_csv ? sum.sectors_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], + stages: sum?.stages_csv ? sum.stages_csv.split(",").map((s) => s.trim()).filter(Boolean) : [], + employees: hc?.value_number != null ? Math.round(hc.value_number) : null, + }; +} +async function loadEntityCoords(env, entityId) { + const r = await env.DB.prepare(`SELECT predicate, value_number FROM facts + WHERE entity_id = ? AND is_current = 1 + AND predicate IN ('person.location.lat','person.location.lng','geo.lat','geo.lng','location.lat','location.lng') + AND value_number IS NOT NULL`).bind(entityId).all(); + let lat = null; + let lng = null; + for (const row of r.results ?? []) { + if (lat == null && (row.predicate.endsWith(".lat") || row.predicate === "geo.lat")) + lat = row.value_number; + if (lng == null && (row.predicate.endsWith(".lng") || row.predicate === "geo.lng")) + lng = row.value_number; + } + return { lat, lng }; +} +export async function loadPersonEntity(env, entityId) { + const ent = await env.DB.prepare(`SELECT id, display_name, kind, status FROM u_entities WHERE id = ?`).bind(entityId).first(); + if (!ent) + return null; + if (ent.kind !== "person") + return null; + if (ent.status === "merged" || ent.status === "soft_deleted") + return null; + const sum = await env.DB.prepare(`SELECT country_iso2, region FROM entity_summary WHERE entity_id = ?`).bind(entityId).first(); + const career = await env.DB.prepare(`SELECT role_title, seniority, department, organization_entity_id, organization_name + FROM career_history + WHERE entity_id = ? + ORDER BY is_current DESC, COALESCE(ended_at, '9999') DESC, started_at DESC + LIMIT 1`).bind(entityId).first(); + let title = career?.role_title ?? null; + if (!title) { + const tf = await env.DB.prepare(`SELECT value_text FROM facts WHERE entity_id = ? AND is_current = 1 AND predicate IN ('person.title','title') AND value_text IS NOT NULL ORDER BY observed_at DESC LIMIT 1`).bind(entityId).first(); + title = tf?.value_text ?? null; + } + let employer = null; + if (career?.organization_entity_id) { + try { + employer = await loadEmployerFacts(env, career.organization_entity_id); + } + catch { /* ignore */ } + } + const coords = await loadEntityCoords(env, entityId).catch(() => ({ lat: null, lng: null })); + return { + id: ent.id, + display_name: ent.display_name, + country_iso2: sum?.country_iso2 ?? null, + region: sum?.region ?? null, + lat: coords.lat, + lng: coords.lng, + title, + seniority: career?.seniority ?? null, + department: career?.department ?? null, + employer_entity_id: career?.organization_entity_id ?? null, + employer_name: employer?.name ?? career?.organization_name ?? null, + employer_country: employer?.country ?? null, + employer_sectors: employer?.sectors ?? [], + employer_stages: employer?.stages ?? [], + employer_employees: employer?.employees ?? null, + }; +} +// --------------------------------------------------------------------------- +// Title similarity (only DB/Env-touching scorer). +// +// Embeddings are cached in persona_title_embeddings + entity_title_embeddings +// keyed by content_hash so the hot path becomes a D1 lookup instead of an +// AI.embed call. AI.embed only fires on cache miss (new persona, new entity, +// or title text change). This mirrors the Vectorize precompute/reuse +// pattern from Task #7 personas while staying on D1. +// --------------------------------------------------------------------------- +async function sha256Hex(input) { + const buf = new TextEncoder().encode(input); + const digest = await crypto.subtle.digest("SHA-256", buf); + return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, "0")).join(""); +} +async function ensureTitleCacheTables(env) { + try { + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_title_embeddings (persona_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS entity_title_embeddings (entity_id TEXT PRIMARY KEY, content_hash TEXT NOT NULL, vector_json TEXT NOT NULL, model TEXT NOT NULL DEFAULT 'bge-base-en-v1.5', updated_at TEXT NOT NULL DEFAULT (datetime('now')))`).run(); + } + catch { /* best-effort */ } +} +async function getOrEmbedTitle(env, scope, id, text) { + const hash = await sha256Hex(text); + const table = scope === "persona" ? "persona_title_embeddings" : "entity_title_embeddings"; + const idCol = scope === "persona" ? "persona_id" : "entity_id"; + try { + const row = await env.DB.prepare(`SELECT vector_json FROM ${table} WHERE ${idCol} = ? AND content_hash = ?`).bind(id, hash).first(); + if (row?.vector_json) { + try { + const v = JSON.parse(row.vector_json); + if (Array.isArray(v) && v.length) + return v; + } + catch { /* fall through to re-embed */ } + } + } + catch { + // Table missing — create it once and continue with embedding path. + await ensureTitleCacheTables(env); + } + if (!env.AI) + return null; + const vec = await aiEmbed(env, text); + if (vec && vec.length) { + try { + await env.DB.prepare(`INSERT INTO ${table} (${idCol}, content_hash, vector_json, updated_at) VALUES (?, ?, ?, datetime('now')) + ON CONFLICT(${idCol}) DO UPDATE SET content_hash=excluded.content_hash, vector_json=excluded.vector_json, updated_at=excluded.updated_at`).bind(id, hash, JSON.stringify(vec)).run(); + } + catch (e) { + console.warn("title embedding cache write failed", scope, id, e.message); + } + } + return vec; +} +async function titleSimilarity(env, personaId, personaText, entityId, entityTitle) { + if (!entityTitle || !personaText) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "missing title" }; + } + if (!env.AI) { + const ov = scoreFunction(entityTitle, [personaText]); + return { value: ov.value, weight: DEFAULT_WEIGHTS.title_sim, reason: `title token overlap (no embed): ${ov.reason}` }; + } + try { + const [pv, ev] = await Promise.all([ + getOrEmbedTitle(env, "persona", personaId, personaText), + getOrEmbedTitle(env, "entity", entityId, entityTitle), + ]); + if (!pv || !ev) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: "embedding unavailable" }; + } + const c = cosine(pv, ev); + return { value: c, weight: DEFAULT_WEIGHTS.title_sim, reason: `title cosine ${c.toFixed(3)} ("${entityTitle}")`, data: { cosine: c } }; + } + catch (e) { + return { value: 0, weight: DEFAULT_WEIGHTS.title_sim, reason: `embed error: ${e.message.slice(0, 60)}` }; + } +} +// --------------------------------------------------------------------------- +// Public API. +// --------------------------------------------------------------------------- +export async function scoreEntityForPersona(env, persona, entity) { + const targets = extractTargets(persona); + const [title_sim, seniority, fnc, industry, company_size, stage, geo] = await Promise.all([ + titleSimilarity(env, persona.id, targets.title_text || targets.titles.join(", "), entity.id, entity.title), + Promise.resolve(scoreSeniority(entity.seniority, targets.seniority)), + Promise.resolve(scoreFunction(entity.department, targets.functions)), + Promise.resolve(scoreIndustry(entity.employer_sectors, targets.industries)), + Promise.resolve(scoreCompanySize(entity.employer_employees, targets.size_min, targets.size_max)), + Promise.resolve(scoreStage(entity.employer_stages, targets.stages)), + Promise.resolve(scoreGeo({ + entityIso2: entity.country_iso2 ?? entity.employer_country, + entityLat: entity.lat, entityLng: entity.lng, + targets: targets.geos, + centerLat: targets.geo_center_lat, centerLng: targets.geo_center_lng, + radiusKm: targets.geo_radius_km, + })), + ]); + const components = { title_sim, seniority, function: fnc, industry, company_size, stage, geo }; + const score = aggregate(components); + const rationale = buildRationale(persona.name, entity.display_name, entity.employer_name, components, score); + return { score, components, rationale }; +} +export async function scoreEntity(env, personaId, entityId) { + const persona = await getPersona(env, personaId); + if (!persona) + return null; + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + // Task #2 budget gate: refuse if AI cap reached (title_sim uses AI.embed). + const b = await assertBudget(env, "ai"); + if (!b.ok) { + await recordMatchJob(env, "score_entity", "halted", { personaId, entityId, reason: b.reason }); + return null; + } + const result = await scoreEntityForPersona(env, persona, entity); + await upsertMatch(env, personaId, entityId, result); + return result; +} +// Task #8: durable job/error log so SLO violations are visible. Created +// on demand by triggering migrations; CREATE TABLE IF NOT EXISTS guards +// against pre-migration calls. +async function ensureJobsTable(env) { + try { + await env.DB.prepare(`CREATE TABLE IF NOT EXISTS persona_match_jobs ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, + status TEXT NOT NULL, + persona_id TEXT, + entity_id TEXT, + details_json TEXT, + created_at TEXT NOT NULL DEFAULT (datetime('now')) + )`).run(); + } + catch { /* best-effort */ } +} +export async function recordMatchJob(env, kind, status, details) { + await ensureJobsTable(env); + try { + await env.DB.prepare(`INSERT INTO persona_match_jobs (id, kind, status, persona_id, entity_id, details_json) + VALUES (?, ?, ?, ?, ?, ?)`).bind(crypto.randomUUID(), kind, status, details.personaId ?? null, details.entityId ?? null, JSON.stringify(details)).run(); + } + catch (e) { + console.warn("recordMatchJob failed", kind, status, e.message); + } +} +async function isCancelled(env, jobId) { + if (!jobId) + return false; + try { + const r = await env.DB.prepare("SELECT status FROM jobs WHERE id = ?").bind(jobId).first(); + return r?.status === "cancelled" || r?.status === "timed_out"; + } + catch { + return false; + } +} +export async function upsertMatch(env, personaId, entityId, result, opts = {}) { + const source = opts.source ?? "auto"; + const evidence = JSON.stringify({ + components: Object.fromEntries(Object.keys(result.components).map((k) => [k, { + value: Number(result.components[k].value.toFixed(4)), + weight: result.components[k].weight, + reason: result.components[k].reason, + }])), + rationale: result.rationale, + weights: DEFAULT_WEIGHTS, + version: MODEL_VERSION, + }); + if (source === "auto") { + await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) + VALUES (?, ?, ?, ?, 'auto', datetime('now'), ?) + ON CONFLICT(persona_id, entity_id) DO UPDATE SET + score = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.score ELSE excluded.score END, + match_evidence_json = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.match_evidence_json ELSE excluded.match_evidence_json END, + last_scored_at = excluded.last_scored_at, + model_version = CASE WHEN persona_entity_matches.source = 'manual' THEN persona_entity_matches.model_version ELSE excluded.model_version END`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); + return; + } + await env.DB.prepare(`INSERT INTO persona_entity_matches (persona_id, entity_id, score, match_evidence_json, source, last_scored_at, model_version) + VALUES (?, ?, ?, ?, 'manual', datetime('now'), ?) + ON CONFLICT(persona_id, entity_id) DO UPDATE SET + score = excluded.score, + match_evidence_json = excluded.match_evidence_json, + source = 'manual', + last_scored_at = excluded.last_scored_at, + model_version = excluded.model_version`).bind(personaId, entityId, result.score, evidence, MODEL_VERSION).run(); +} +export async function scoreEntityAcrossPersonas(env, entityId, opts = {}) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return { scored: 0, errors: 0, halted: false }; + const r = await env.DB.prepare(`SELECT * FROM personas WHERE deleted_at IS NULL AND status = 'active'`).all(); + let scored = 0; + let errors = 0; + let halted = false; + for (const p of r.results ?? []) { + // Task #2: budget + cancellation enforcement per item. + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "score_across_personas", "halted", { entityId, scored, errors, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "score_across_personas", "cancelled", { entityId, scored, errors }); + break; + } + try { + const res = await scoreEntityForPersona(env, p, entity); + await upsertMatch(env, p.id, entityId, res); + scored += 1; + } + catch (e) { + errors += 1; + console.warn("scoreEntityAcrossPersonas item failed", p.id, entityId, e.message); + } + } + return { scored, errors, halted }; +} +export async function scoreBatch(env, personaId, opts = {}) { + const persona = await getPersona(env, personaId); + if (!persona) + return { scored: 0, errors: 0, pages: 0, halted: false }; + const batchSize = Math.min(Math.max(1, opts.batchSize ?? 100), 500); + // maxEntities = null (default) means "process every active person + // entity" — the task requires create/edit dispatch covers all + // entities, not a hardcoded cap. Operators can pass a number when + // they want to bound a manual run. + const maxEntities = opts.maxEntities ?? null; + let offset = 0; + let scored = 0; + let errors = 0; + let pages = 0; + let halted = false; + for (;;) { + if (maxEntities != null && scored + errors >= maxEntities) + break; + // Task #2: budget + cancellation check per page (cheap, bounded). + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "score_batch", "halted", { personaId, scored, errors, pages, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "score_batch", "cancelled", { personaId, scored, errors, pages }); + break; + } + // Task #3: dispatch through the kind plugin so each persona kind + // selects its own candidate pool (e.g. investor_person filters + // entity_roles.role IN ('investor','vc','gp','partner_at_firm')). + // Note: explicit .js extension here is required by tsconfig.test.json's + // NodeNext moduleResolution. The wrangler build / typecheck doesn't + // care; this is purely to unblock `pnpm test`. + const { getPluginFor } = await import("./personas/kinds/index.js"); + const plugin = getPluginFor(persona.kind); + const filter = plugin.defaultEntityFilter(persona, { limit: batchSize, offset }); + const r = await env.DB.prepare(filter.sql).bind(...filter.binds).all(); + const ids = (r.results ?? []).map((x) => x.id); + if (!ids.length) + break; + pages += 1; + for (const id of ids) { + try { + // Task #3: delegate to the kind plugin so bespoke matchers + // (investor_firm structural, venture_partner subtype, etc.) + // get the chance to override scoring. The generic plugin's + // scoreEntity returns the person-graph score for person + // targets and null for fund/company targets — when null, we + // persist a deterministic structural-match row at score 50 + // so non-person kinds still surface candidates in the UI. + let res = await plugin.scoreEntity(env, persona, id); + if (!res) + res = buildStructuralFallback(plugin.explainMatch(id)); + await upsertMatch(env, personaId, id, res); + scored += 1; + } + catch (e) { + errors += 1; + console.warn("scoreBatch item failed", personaId, id, e.message); + } + } + if (ids.length < batchSize) + break; + offset += batchSize; + } + if (errors > 0 && !halted) { + await recordMatchJob(env, "score_batch", "ok", { personaId, scored, errors, pages }); + } + return { scored, errors, pages, halted }; +} +export async function refreshStaleMatches(env, opts = {}) { + const staleDays = Math.max(1, opts.staleDays ?? 30); + const limit = Math.min(Math.max(1, opts.limit ?? 500), 5000); + const r = await env.DB.prepare(`SELECT persona_id, entity_id FROM persona_entity_matches + WHERE source = 'auto' AND datetime(last_scored_at) < datetime('now', ?) + ORDER BY last_scored_at ASC LIMIT ?`).bind(`-${staleDays} days`, limit).all(); + let refreshed = 0; + let errors = 0; + let halted = false; + for (const row of r.results ?? []) { + // Task #2: per-item budget + cancellation gate. + const b = await assertBudget(env, "ai"); + if (!b.ok) { + halted = true; + await recordMatchJob(env, "refresh_stale", "halted", { refreshed, errors, reason: b.reason }); + break; + } + if (await isCancelled(env, opts.jobId ?? null)) { + halted = true; + await recordMatchJob(env, "refresh_stale", "cancelled", { refreshed, errors }); + break; + } + try { + const res = await scoreEntity(env, row.persona_id, row.entity_id); + if (res) + refreshed += 1; + } + catch (e) { + errors += 1; + console.warn("refreshStaleMatches item failed", row.persona_id, row.entity_id, e.message); + } + } + return { refreshed, errors, halted }; +} +export async function listCandidates(env, personaId, opts) { + const minScore = Math.max(0, Math.min(1, opts.minScore ?? 0)); + const limit = Math.min(Math.max(1, opts.limit ?? 50), 500); + const offset = Math.max(0, opts.offset ?? 0); + const r = await env.DB.prepare(`SELECT pem.persona_id, pem.entity_id, pem.score, pem.source, pem.last_scored_at, pem.model_version, pem.match_evidence_json, + ue.display_name AS entity_name, ue.primary_domain AS entity_domain, + es.country_iso2 AS entity_country + FROM persona_entity_matches pem + JOIN u_entities ue ON ue.id = pem.entity_id + LEFT JOIN entity_summary es ON es.entity_id = pem.entity_id + WHERE pem.persona_id = ? AND pem.score >= ? + ORDER BY pem.score DESC, pem.last_scored_at DESC + LIMIT ? OFFSET ?`).bind(personaId, minScore, limit, offset).all(); + return (r.results ?? []).map((row) => { + let components = {}; + let rationale = ""; + if (row.match_evidence_json) { + try { + const j = JSON.parse(row.match_evidence_json); + if (j.components) + components = j.components; + if (typeof j.rationale === "string") + rationale = j.rationale; + } + catch { /* ignore */ } + } + return { + persona_id: row.persona_id, + entity_id: row.entity_id, + score: row.score, + source: row.source, + last_scored_at: row.last_scored_at, + model_version: row.model_version, + components, + rationale, + entity_name: row.entity_name, + entity_domain: row.entity_domain, + entity_country: row.entity_country, + }; + }); +} diff --git a/apps/worker/test-dist-q/services/personaMatchingScorers.js b/apps/worker/test-dist-q/services/personaMatchingScorers.js new file mode 100644 index 00000000..0d68be9d --- /dev/null +++ b/apps/worker/test-dist-q/services/personaMatchingScorers.js @@ -0,0 +1,366 @@ +// Task #8: Pure scoring primitives for the persona ↔ entity matcher. +// +// This module is intentionally free of D1 / Env / AI imports so it can +// be unit-tested in node:test (see test/personaMatching.test.mjs). The +// orchestrator (services/personaMatching.ts) wires these primitives to +// the database, embeddings, and workflow dispatch. +export const MODEL_VERSION = "v1"; +export const DEFAULT_WEIGHTS = { + title_sim: 0.25, + seniority: 0.15, + function: 0.15, + industry: 0.15, + company_size: 0.10, + stage: 0.10, + geo: 0.10, +}; +const SENIORITY_LADDER = [ + "ic", "analyst", "associate", "manager", "principal", + "director", "vp", "svp", "cxo", "founder", "partner", +]; +const SENIORITY_INDEX = Object.fromEntries(SENIORITY_LADDER.map((s, i) => [s, i])); +const STAGE_LADDER = [ + "pre_seed", "seed", "series_a", "series_b", "series_c", + "series_d", "growth", "late", "public", +]; +const STAGE_INDEX = Object.fromEntries(STAGE_LADDER.map((s, i) => [s, i])); +const INDUSTRY_PARENTS = { + fintech: ["finance"], insurtech: ["finance"], wealthtech: ["finance"], + proptech: ["realestate"], regtech: ["finance", "compliance"], + edtech: ["education"], healthtech: ["healthcare"], + biotech: ["healthcare", "lifesciences"], medtech: ["healthcare"], + cleantech: ["energy"], climatetech: ["energy"], + saas: ["software"], devtools: ["software"], paas: ["software"], + martech: ["marketing"], adtech: ["marketing"], + agtech: ["agriculture"], foodtech: ["food"], + legaltech: ["legal"], hrtech: ["hr"], +}; +const CONTINENT = { + us: "na", ca: "na", mx: "na", + gb: "eu", de: "eu", fr: "eu", es: "eu", it: "eu", nl: "eu", se: "eu", ch: "eu", ie: "eu", pl: "eu", pt: "eu", be: "eu", at: "eu", dk: "eu", no: "eu", fi: "eu", + cn: "as", jp: "as", in: "as", sg: "as", kr: "as", hk: "as", il: "as", ae: "as", + br: "sa", ar: "sa", cl: "sa", co: "sa", + au: "oc", nz: "oc", + za: "af", ng: "af", ke: "af", eg: "af", +}; +// Rough country centroids (lat, lng) used as a fallback when an entity +// has an ISO2 country but no precise coordinates. Only the most common +// targets are populated; absent entries skip the Haversine path and +// fall back to ISO2 + continent. +const COUNTRY_CENTROID = { + us: [39.8, -98.6], ca: [56.1, -106.3], mx: [23.6, -102.5], + gb: [54.0, -2.0], de: [51.2, 10.4], fr: [46.6, 2.2], + es: [40.4, -3.7], it: [41.9, 12.5], nl: [52.1, 5.3], + se: [60.1, 18.6], ch: [46.8, 8.2], ie: [53.1, -7.7], + cn: [35.9, 104.2], jp: [36.2, 138.2], in: [20.6, 78.9], + sg: [1.35, 103.8], kr: [35.9, 127.8], il: [31.0, 34.9], + br: [-14.2, -51.9], au: [-25.3, 133.8], nz: [-40.9, 174.9], + za: [-30.6, 22.9], ng: [9.1, 8.7], +}; +// --------------------------------------------------------------------------- +// Cosine similarity (exported for the embedding-driven title_sim). +// --------------------------------------------------------------------------- +export function cosine(a, b) { + if (a.length !== b.length || !a.length) + return 0; + let dot = 0, na = 0, nb = 0; + for (let i = 0; i < a.length; i++) { + dot += a[i] * b[i]; + na += a[i] * a[i]; + nb += b[i] * b[i]; + } + if (!na || !nb) + return 0; + return Math.max(0, dot / (Math.sqrt(na) * Math.sqrt(nb))); +} +// --------------------------------------------------------------------------- +// Component scorers (each returns a value in [0, 1]). +// --------------------------------------------------------------------------- +export function scoreSeniority(entity, targets) { + if (!entity || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: "no seniority data" }; + } + const e = entity.toLowerCase().trim(); + const ei = SENIORITY_INDEX[e]; + if (ei === undefined) { + return { value: 0, weight: DEFAULT_WEIGHTS.seniority, reason: `unknown seniority "${entity}"` }; + } + let best = 0; + let bestTarget = ""; + for (const t of targets) { + const ti = SENIORITY_INDEX[t.toLowerCase().trim()]; + if (ti === undefined) + continue; + const d = Math.abs(ei - ti); + let v = 0; + if (d === 0) + v = 1.0; + else if (d === 1) + v = 0.6; + else if (d === 2) + v = 0.2; + if (v > best) { + best = v; + bestTarget = t; + } + } + return { + value: best, weight: DEFAULT_WEIGHTS.seniority, + reason: best === 1 ? `exact seniority match (${entity})` + : best > 0 ? `seniority ${entity} ≈ target ${bestTarget}` + : `seniority ${entity} too far from targets`, + }; +} +function stem(token) { + let t = token.toLowerCase().replace(/[^a-z]/g, ""); + if (t.endsWith("ing") && t.length > 5) + t = t.slice(0, -3); + else if (t.endsWith("ed") && t.length > 4) + t = t.slice(0, -2); + else if (t.endsWith("es") && t.length > 4) + t = t.slice(0, -2); + else if (t.endsWith("s") && t.length > 3) + t = t.slice(0, -1); + return t; +} +function tokenSet(s) { + if (!s) + return new Set(); + return new Set(s.split(/[\s,/&-]+/).map(stem).filter((t) => t.length > 1)); +} +export function scoreFunction(entityDept, targets) { + if (!entityDept || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.function, reason: "no function/dept data" }; + } + const ent = tokenSet(entityDept); + let best = 0; + let bestTarget = ""; + for (const t of targets) { + const tgt = tokenSet(t); + if (!tgt.size) + continue; + let hits = 0; + for (const tok of tgt) + if (ent.has(tok)) + hits++; + const jacc = hits / Math.max(1, new Set([...ent, ...tgt]).size); + if (jacc > best) { + best = jacc; + bestTarget = t; + } + } + return { + value: Math.min(1, best * 1.5), + weight: DEFAULT_WEIGHTS.function, + reason: best > 0 ? `function "${entityDept}" overlaps "${bestTarget}"` : `function "${entityDept}" no overlap`, + }; +} +export function scoreIndustry(entityIndustries, targets) { + if (!entityIndustries.length || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.industry, reason: "no industry data" }; + } + const tset = new Set(targets.map((t) => t.toLowerCase())); + let best = 0; + let bestNote = ""; + for (const ei of entityIndustries) { + const e = ei.toLowerCase(); + if (tset.has(e)) { + best = 1.0; + bestNote = `industry "${ei}" matches target`; + break; + } + const parents = INDUSTRY_PARENTS[e] ?? []; + for (const p of parents) { + if (tset.has(p)) { + if (best < 0.7) { + best = 0.7; + bestNote = `industry "${ei}" ⊂ target "${p}"`; + } + } + } + } + if (!bestNote) + bestNote = `industries [${entityIndustries.join(",")}] don't match targets`; + return { value: best, weight: DEFAULT_WEIGHTS.industry, reason: bestNote }; +} +export function scoreCompanySize(emp, minE, maxE) { + if (emp == null || (minE == null && maxE == null)) { + return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: "company size unknown" }; + } + const lo = minE ?? 0; + const hi = maxE ?? Number.POSITIVE_INFINITY; + if (emp >= lo && emp <= hi) { + return { value: 1.0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} in target [${lo}, ${maxE ?? "∞"}]` }; + } + const dLo = emp < lo ? (lo - emp) / Math.max(1, lo) : 0; + const dHi = emp > hi ? (emp - hi) / Math.max(1, hi) : 0; + const d = Math.max(dLo, dHi); + if (d <= 0.5) + return { value: 0.5, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} adjacent to target` }; + return { value: 0, weight: DEFAULT_WEIGHTS.company_size, reason: `headcount ${emp} outside target [${lo}, ${maxE ?? "∞"}]` }; +} +export function scoreStage(entityStages, targets) { + if (!entityStages.length || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.stage, reason: "stage unknown" }; + } + let best = 0; + let note = ""; + for (const e of entityStages) { + const ei = STAGE_INDEX[e.toLowerCase().replace(/[-\s]/g, "_")]; + if (ei === undefined) + continue; + for (const t of targets) { + const ti = STAGE_INDEX[t.toLowerCase().replace(/[-\s]/g, "_")]; + if (ti === undefined) + continue; + const d = Math.abs(ei - ti); + let v = 0; + if (d === 0) + v = 1.0; + else if (d === 1) + v = 0.6; + if (v > best) { + best = v; + note = d === 0 ? `stage ${e} matches target` : `stage ${e} adjacent to ${t}`; + } + } + } + return { value: best, weight: DEFAULT_WEIGHTS.stage, reason: note || `stages [${entityStages.join(",")}] don't match targets` }; +} +// --------------------------------------------------------------------------- +// Geo: Haversine + exponential decay when coordinates are present; ISO2 +// + continent fallback otherwise. +// --------------------------------------------------------------------------- +const EARTH_KM = 6371; +export function haversineKm(lat1, lng1, lat2, lng2) { + const toRad = (x) => (x * Math.PI) / 180; + const dLat = toRad(lat2 - lat1); + const dLng = toRad(lng2 - lng1); + const a = Math.sin(dLat / 2) ** 2 + + Math.cos(toRad(lat1)) * Math.cos(toRad(lat2)) * Math.sin(dLng / 2) ** 2; + return 2 * EARTH_KM * Math.asin(Math.min(1, Math.sqrt(a))); +} +export function scoreGeo(input) { + const { entityIso2, targets } = input; + const hasCenter = input.centerLat != null && input.centerLng != null && (input.radiusKm ?? 0) > 0; + // Coordinate path: Haversine + exp(-d/radius) decay. + let entLat = input.entityLat ?? null; + let entLng = input.entityLng ?? null; + if ((entLat == null || entLng == null) && entityIso2) { + const c = COUNTRY_CENTROID[entityIso2.toLowerCase()]; + if (c) { + entLat = c[0]; + entLng = c[1]; + } + } + if (hasCenter && entLat != null && entLng != null) { + const d = haversineKm(input.centerLat, input.centerLng, entLat, entLng); + const v = Math.max(0, Math.min(1, Math.exp(-d / Math.max(1, input.radiusKm)))); + return { + value: v, + weight: DEFAULT_WEIGHTS.geo, + reason: `geo ${d.toFixed(0)}km from persona center (radius ${input.radiusKm}km) ⇒ ${v.toFixed(2)}`, + data: { distance_km: d, radius_km: input.radiusKm }, + }; + } + // ISO2 fallback. + if (!entityIso2 || !targets.length) { + return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: "geo unknown" }; + } + const e = entityIso2.toLowerCase(); + const t = targets.map((x) => x.toLowerCase()); + if (t.includes(e)) { + return { value: 1.0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} matches target` }; + } + const ec = CONTINENT[e]; + if (ec) { + for (const tc of t) + if (CONTINENT[tc] === ec) { + return { value: 0.5, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} shares region with target ${tc.toUpperCase()}` }; + } + } + return { value: 0, weight: DEFAULT_WEIGHTS.geo, reason: `geo ${entityIso2} not in targets [${targets.join(",")}]` }; +} +// --------------------------------------------------------------------------- +// Aggregation + rationale. +// --------------------------------------------------------------------------- +export function aggregate(components) { + let sum = 0; + let wsum = 0; + for (const k of Object.keys(components)) { + const c = components[k]; + sum += c.value * c.weight; + wsum += c.weight; + } + return wsum > 0 ? sum / wsum : 0; +} +export function buildRationale(personaName, entityName, employerName, components, score) { + const pct = Math.round(score * 100); + const top = Object.entries(components) + .map(([k, c]) => ({ k, contribution: c.value * c.weight, reason: c.reason })) + .sort((a, b) => b.contribution - a.contribution) + .slice(0, 3) + .map((x) => x.reason) + .join("; "); + const who = entityName ?? "entity"; + const where = employerName ? ` at ${employerName}` : ""; + return `${who}${where} scores ${pct}% against persona "${personaName}". Top drivers: ${top}.`; +} +function arrFromJson(s) { + if (!s) + return []; + try { + const v = JSON.parse(s); + return Array.isArray(v) ? v.filter((x) => typeof x === "string") : []; + } + catch { + return []; + } +} +function objFromJson(s) { + if (!s) + return {}; + try { + const v = JSON.parse(s); + return v && typeof v === "object" && !Array.isArray(v) ? v : {}; + } + catch { + return {}; + } +} +export function extractTargets(row) { + const hard = objFromJson(row.hard_filters_json); + const stagesFromHard = Array.isArray(hard.stages) ? hard.stages.filter((x) => typeof x === "string") + : Array.isArray(hard.target_stage) ? hard.target_stage.filter((x) => typeof x === "string") + : []; + const center = hard.geo_center && typeof hard.geo_center === "object" ? hard.geo_center : {}; + const titles = arrFromJson(row.buyer_titles_json); + const seniority = arrFromJson(row.buyer_seniority_json); + const functions = arrFromJson(row.buyer_departments_json); + const industries = arrFromJson(row.industries_json); + const geos = arrFromJson(row.geos_json); + // Task #8 spec: title_sim must use ONLY structured persona target + // fields — no long-form notes (thesis, free text). Embedding here is + // restricted to titles + seniority + function so the component score + // stays explainable and reproducible. + const title_text = [ + titles.join(", "), + seniority.length ? `Seniority: ${seniority.join(", ")}` : "", + functions.length ? `Function: ${functions.join(", ")}` : "", + ].filter(Boolean).join(". "); + return { + title_text, + titles, + seniority, + functions, + industries, + size_min: row.size_min ?? null, + size_max: row.size_max ?? null, + stages: stagesFromHard, + geos, + geo_center_lat: typeof center.lat === "number" ? center.lat : null, + geo_center_lng: typeof center.lng === "number" ? center.lng : null, + geo_radius_km: typeof center.radius_km === "number" ? center.radius_km + : typeof hard.radius_km === "number" ? hard.radius_km : null, + }; +} diff --git a/apps/worker/test-dist-q/services/personas/kinds/_generic.js b/apps/worker/test-dist-q/services/personas/kinds/_generic.js new file mode 100644 index 00000000..7af6e0ef --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/_generic.js @@ -0,0 +1,45 @@ +// Task #3: Generic kind plugin used as the default for kinds that +// don't ship a bespoke matcher. Drives candidate selection from the +// taxonomy's `targets` (entity kind) + `roles` (entity_roles.role IN) +// and delegates scoring to the existing PersonaMatchingService scorer +// for person targets. Company/fund targets currently fall back to the +// legacy persona_matches/accounts/buyers code path via the dispatcher. +import { loadPersonEntity, scoreEntityForPersona as scorePersonForPersona } from "../../personaMatching"; +import { KINDS } from "./taxonomy"; +export function makeGenericPlugin(kind) { + const def = KINDS[kind]; + return { + kind, + defaultEntityFilter(_persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + const binds = [def.targets]; + let sql = `SELECT DISTINCT e.id FROM u_entities e`; + if (def.roles.length) { + sql += ` JOIN entity_roles r ON r.entity_id = e.id`; + } + sql += ` WHERE e.kind = ? AND e.status = 'active'`; + if (def.roles.length) { + sql += ` AND r.role IN (${def.roles.map(() => "?").join(",")})`; + binds.push(...def.roles); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(env, persona, entityId) { + // Default behavior: only person targets are scored via the + // graph scorer. Fund/company targets are out of scope for the + // person-graph matcher and return null so the caller skips them. + if (def.targets !== "person") + return null; + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scorePersonForPersona(env, persona, entity); + }, + explainMatch(entityId) { + return `kind=${kind} target=${def.targets}${def.roles.length ? " roles=" + def.roles.join("|") : ""} entity=${entityId}`; + }, + }; +} diff --git a/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js b/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js new file mode 100644 index 00000000..a56a2ac2 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/academic_researcher.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=academic_researcher. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AcademicResearcherPlugin = makeGenericPlugin("academic_researcher"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/account_company.js b/apps/worker/test-dist-q/services/personas/kinds/account_company.js new file mode 100644 index 00000000..8072a299 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/account_company.js @@ -0,0 +1,17 @@ +// Task #3: account_company kind plugin (legacy "account" kind). +// +// Sales-side accounts live in the legacy `accounts` table and are +// scored via personas/score.ts + persona_matches (Task #46), not via +// the u_entities person graph. This plugin exists so the dispatcher +// can identify the kind, but defaultEntityFilter returns an empty +// candidate set — the legacy code path in routes/personas.ts owns +// account rescoring end-to-end. +export const accountCompanyPlugin = { + kind: "account_company", + defaultEntityFilter(_persona, _opts) { + // Legacy path: accounts are not in u_entities for persona matching. + return { sql: `SELECT id FROM u_entities WHERE 1 = 0`, binds: [] }; + }, + async scoreEntity(_env, _persona, _entityId) { return null; }, + explainMatch(entityId) { return `account_company (legacy accounts table): entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/acquirer.js b/apps/worker/test-dist-q/services/personas/kinds/acquirer.js new file mode 100644 index 00000000..72f5b2c9 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/acquirer.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=acquirer. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AcquirerPlugin = makeGenericPlugin("acquirer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js b/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js new file mode 100644 index 00000000..d4d9efde --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/angel_individual.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=angel_individual. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const AngelIndividualPlugin = makeGenericPlugin("angel_individual"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js b/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js new file mode 100644 index 00000000..40cbde98 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/beta_tester.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=beta_tester. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const BetaTesterPlugin = makeGenericPlugin("beta_tester"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js b/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js new file mode 100644 index 00000000..6f8f008e --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/buyer_person.js @@ -0,0 +1,37 @@ +// Task #3: buyer_person kind plugin (legacy "buyer" kind). +// +// Combines: (a) the new u_entities person-graph filter on role IN +// ('buyer','decision_maker','champion'), (b) delegation to the +// person-graph scorer. The legacy `buyers` table is still rescored +// by personas/rescore.ts under the hood; this plugin only governs +// the new u_entities-backed candidate list. +import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; +const ROLES = ["buyer", "decision_maker", "champion"]; +export const buyerPersonPlugin = { + kind: "buyer_person", + defaultEntityFilter(_persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + const ph = ROLES.map(() => "?").join(","); + return { + // No role filter when entity_roles lacks buyer-flavored rows yet + // — fall back to all active person entities. Matches the legacy + // behavior so existing 'buyer' personas don't regress to empty. + sql: `SELECT e.id FROM u_entities e + WHERE e.kind = 'person' AND e.status = 'active' + AND ( + EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id AND r.role IN (${ph})) + OR NOT EXISTS (SELECT 1 FROM entity_roles r WHERE r.entity_id = e.id) + ) + ORDER BY e.id LIMIT ? OFFSET ?`, + binds: [...ROLES, limit, offset], + }; + }, + async scoreEntity(env, persona, entityId) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scoreEntityForPersona(env, persona, entity); + }, + explainMatch(entityId) { return `buyer_person match: entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js b/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js new file mode 100644 index 00000000..3eb5cc89 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/channel_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=channel_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ChannelPartnerPlugin = makeGenericPlugin("channel_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js b/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js new file mode 100644 index 00000000..c3786ce0 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/co_founder_match.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=co_founder_match. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const CoFounderMatchPlugin = makeGenericPlugin("co_founder_match"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/competitor.js b/apps/worker/test-dist-q/services/personas/kinds/competitor.js new file mode 100644 index 00000000..3801e793 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/competitor.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=competitor. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const CompetitorPlugin = makeGenericPlugin("competitor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/design_partner.js b/apps/worker/test-dist-q/services/personas/kinds/design_partner.js new file mode 100644 index 00000000..04200099 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/design_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=design_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const DesignPartnerPlugin = makeGenericPlugin("design_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js b/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js new file mode 100644 index 00000000..ecebae4a --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/engineering_hire.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=engineering_hire. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const EngineeringHirePlugin = makeGenericPlugin("engineering_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js b/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js new file mode 100644 index 00000000..a45b4674 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/executive_hire.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=executive_hire. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ExecutiveHirePlugin = makeGenericPlugin("executive_hire"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/founder.js b/apps/worker/test-dist-q/services/personas/kinds/founder.js new file mode 100644 index 00000000..6ef92773 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/founder.js @@ -0,0 +1,68 @@ +// Task #3: founder kind plugin. +// +// Surfaces hint fields `founded_count`, `prior_exits`, `domain_expertise`. +// Match: entities with role IN ('founder','ceo','co_founder'). Hint +// numerics are validated by the form; we re-validate here so a stale +// or hand-edited persona can't crash the scorer. +import { loadPersonEntity, scoreEntityForPersona } from "../../personaMatching"; +const ROLES = ["founder", "ceo", "co_founder"]; +function readHint(persona, field) { + if (!persona.hard_filters_json) + return null; + try { + const j = JSON.parse(persona.hard_filters_json); + const v = j?.hints?.[field]; + return v == null ? null : String(v); + } + catch { + return null; + } +} +function splitCsv(v) { + return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; +} +export const founderPlugin = { + kind: "founder", + defaultEntityFilter(persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + // Hints become hard predicates so they actually narrow the + // candidate set (not just decorate the UI). Roles in entity_roles + // act as a tag space — domain expertise becomes a 'domain:' + // tag, prior_exits / founded_count map to 'exits:N+' / 'founded:N+' + // synthetic tags that the enrichment pipeline emits per founder. + const domains = splitCsv(readHint(persona, "domain_expertise")); + const foundedMin = parseInt(readHint(persona, "founded_count") ?? "", 10); + const exitsMin = parseInt(readHint(persona, "prior_exits") ?? "", 10); + const binds = []; + const rolePh = ROLES.map(() => "?").join(","); + let sql = `SELECT DISTINCT e.id FROM u_entities e + JOIN entity_roles r ON r.entity_id = e.id + WHERE e.kind = 'person' AND e.status = 'active' + AND r.role IN (${rolePh})`; + binds.push(...ROLES); + if (domains.length) { + const ph = domains.map(() => "?").join(","); + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rd WHERE rd.entity_id = e.id AND rd.role IN (${ph}))`; + binds.push(...domains.map((d) => `domain:${d}`)); + } + if (Number.isFinite(foundedMin) && foundedMin > 0) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rf WHERE rf.entity_id = e.id AND rf.role = ?)`; + binds.push(`founded:${foundedMin}+`); + } + if (Number.isFinite(exitsMin) && exitsMin > 0) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles re WHERE re.entity_id = e.id AND re.role = ?)`; + binds.push(`exits:${exitsMin}+`); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(env, persona, entityId) { + const entity = await loadPersonEntity(env, entityId); + if (!entity) + return null; + return await scoreEntityForPersona(env, persona, entity); + }, + explainMatch(entityId) { return `founder match: entity=${entityId}`; }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js b/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js new file mode 100644 index 00000000..6bcd934d --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/fractional_executive.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=fractional_executive. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const FractionalExecutivePlugin = makeGenericPlugin("fractional_executive"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js b/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js new file mode 100644 index 00000000..0c548ec6 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/government_grant_officer.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=government_grant_officer. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const GovernmentGrantOfficerPlugin = makeGenericPlugin("government_grant_officer"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/index.js b/apps/worker/test-dist-q/services/personas/kinds/index.js new file mode 100644 index 00000000..ed940044 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/index.js @@ -0,0 +1,71 @@ +// Task #3: persona-kind plugin registry + dispatcher entrypoint. +// +// The PersonaMatchingService reads the persona's kind, calls +// getPluginFor(kind), and delegates to defaultEntityFilter / scoreEntity. +// Kinds without a bespoke plugin file fall back to the generic plugin +// driven by the taxonomy's `roles` array (see _generic.ts). +import { ALL_KIND_KEYS, KINDS, resolveKind } from "./taxonomy"; +import { investorPersonPlugin } from "./investor_person"; +import { investorFirmPlugin } from "./investor_firm"; +import { venturePartnerPlugin } from "./venture_partner"; +import { founderPlugin } from "./founder"; +import { accountCompanyPlugin } from "./account_company"; +import { buyerPersonPlugin } from "./buyer_person"; +import { AngelIndividualPlugin } from "./angel_individual"; +import { LimitedPartnerPlugin } from "./limited_partner"; +import { CoFounderMatchPlugin } from "./co_founder_match"; +import { ExecutiveHirePlugin } from "./executive_hire"; +import { EngineeringHirePlugin } from "./engineering_hire"; +import { FractionalExecutivePlugin } from "./fractional_executive"; +import { ChannelPartnerPlugin } from "./channel_partner"; +import { IntegrationPartnerPlugin } from "./integration_partner"; +import { DesignPartnerPlugin } from "./design_partner"; +import { BetaTesterPlugin } from "./beta_tester"; +import { JournalistAnalystPlugin } from "./journalist_analyst"; +import { ThoughtLeaderPlugin } from "./thought_leader"; +import { AcademicResearcherPlugin } from "./academic_researcher"; +import { GovernmentGrantOfficerPlugin } from "./government_grant_officer"; +import { RegulatorPlugin } from "./regulator"; +import { PolicyAdvisorPlugin } from "./policy_advisor"; +import { ServiceProviderPlugin } from "./service_provider"; +import { AcquirerPlugin } from "./acquirer"; +import { CompetitorPlugin } from "./competitor"; +const REGISTRY_MAP = { + account_company: accountCompanyPlugin, + buyer_person: buyerPersonPlugin, + investor_person: investorPersonPlugin, + investor_firm: investorFirmPlugin, + venture_partner: venturePartnerPlugin, + founder: founderPlugin, + angel_individual: AngelIndividualPlugin, + limited_partner: LimitedPartnerPlugin, + co_founder_match: CoFounderMatchPlugin, + executive_hire: ExecutiveHirePlugin, + engineering_hire: EngineeringHirePlugin, + fractional_executive: FractionalExecutivePlugin, + channel_partner: ChannelPartnerPlugin, + integration_partner: IntegrationPartnerPlugin, + design_partner: DesignPartnerPlugin, + beta_tester: BetaTesterPlugin, + journalist_analyst: JournalistAnalystPlugin, + thought_leader: ThoughtLeaderPlugin, + academic_researcher: AcademicResearcherPlugin, + government_grant_officer: GovernmentGrantOfficerPlugin, + regulator: RegulatorPlugin, + policy_advisor: PolicyAdvisorPlugin, + service_provider: ServiceProviderPlugin, + acquirer: AcquirerPlugin, + competitor: CompetitorPlugin, +}; +// Sanity check: every declared kind has a plugin file. If a future +// kind is added to taxonomy without a wrapper, surface it at boot. +for (const k of ALL_KIND_KEYS) { + if (!REGISTRY_MAP[k]) + throw new Error(`missing plugin file for kind=${k}`); +} +export function getPluginFor(rawKind) { + const k = resolveKind(rawKind) ?? "account_company"; + return REGISTRY_MAP[k]; +} +export { KINDS, ALL_KIND_KEYS, resolveKind }; +export * from "./taxonomy"; diff --git a/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js b/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js new file mode 100644 index 00000000..194ec5ec --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/integration_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=integration_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const IntegrationPartnerPlugin = makeGenericPlugin("integration_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js b/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js new file mode 100644 index 00000000..c4af4a4f --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/investor_firm.js @@ -0,0 +1,64 @@ +// Task #3: investor_firm kind plugin. +// +// Targets entities where kind='fund' (or 'firm' legacy) AND +// entity_roles.role='investor_firm'. The person-graph scorer doesn't +// apply to funds directly; matching here is structural (role + kind +// + AUM/stage hints) and we expose a deterministic surface so the +// dispatcher can present candidates even without per-entity scoring. +// Read a hint value from hard_filters_json.hints.. +function readHint(persona, field) { + if (!persona.hard_filters_json) + return null; + try { + const j = JSON.parse(persona.hard_filters_json); + const v = j?.hints?.[field]; + return typeof v === "string" && v ? v : null; + } + catch { + return null; + } +} +// Split a comma-separated hint value into trimmed tokens. +function splitCsv(v) { + return v ? v.split(",").map((s) => s.trim()).filter(Boolean) : []; +} +export const investorFirmPlugin = { + kind: "investor_firm", + defaultEntityFilter(persona, opts) { + const limit = Math.max(1, Math.min(opts?.limit ?? 100, 500)); + const offset = Math.max(0, opts?.offset ?? 0); + // Roles in entity_roles act as a tag space. When the persona + // specifies aum_band or stage_focus hints, we require that the + // fund carries the corresponding tag-role (e.g. 'aum:$1B-$5B' or + // 'stage:seed'). This way the hints actually narrow the candidate + // set rather than just decorating the UI. + const aum = readHint(persona, "aum_band"); // e.g. "$1B-$5B" + const stages = splitCsv(readHint(persona, "stage_focus")); // e.g. ["seed","series_a"] + const binds = []; + let sql = `SELECT DISTINCT e.id FROM u_entities e + JOIN entity_roles r ON r.entity_id = e.id + WHERE e.status = 'active' + AND e.kind IN ('fund','firm') + AND r.role = 'investor_firm'`; + if (aum) { + sql += ` AND EXISTS (SELECT 1 FROM entity_roles ra WHERE ra.entity_id = e.id AND ra.role = ?)`; + binds.push(`aum:${aum}`); + } + if (stages.length) { + const ph = stages.map(() => "?").join(","); + sql += ` AND EXISTS (SELECT 1 FROM entity_roles rs WHERE rs.entity_id = e.id AND rs.role IN (${ph}))`; + binds.push(...stages.map((s) => `stage:${s}`)); + } + sql += ` ORDER BY e.id LIMIT ? OFFSET ?`; + binds.push(limit, offset); + return { sql, binds }; + }, + async scoreEntity(_env, _persona, _entityId) { + // Structural-only match for funds — no per-entity scoring at this + // tier. Dispatcher treats null as "candidate present, score 0.5". + return null; + }, + explainMatch(entityId) { + return `investor_firm structural match: entity=${entityId} role=investor_firm kind=fund|firm`; + }, +}; diff --git a/apps/worker/test-dist-q/services/personas/kinds/investor_person.js b/apps/worker/test-dist-q/services/personas/kinds/investor_person.js new file mode 100644 index 00000000..dfec68e4 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/investor_person.js @@ -0,0 +1,14 @@ +// Task #3: investor_person kind plugin. +// +// Acceptance criteria: matches entities where type='person' AND +// entity_roles.role IN ('investor','vc','gp','partner_at_firm'), +// then filters by firm size / stage / sector criteria via the +// existing person-graph scorer (which already considers employer +// sectors / stages / employees from career_history + entity_summary). +import { makeGenericPlugin } from "./_generic"; +// The generic plugin's defaultEntityFilter already picks up the +// taxonomy's role list ['investor','vc','gp','partner_at_firm'] and +// the person scorer already weighs employer sector / stage / size. +// We export it under a stable name so the registry can swap in a +// bespoke implementation later without touching the registry wiring. +export const investorPersonPlugin = makeGenericPlugin("investor_person"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js b/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js new file mode 100644 index 00000000..086a2293 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/journalist_analyst.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=journalist_analyst. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const JournalistAnalystPlugin = makeGenericPlugin("journalist_analyst"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js b/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js new file mode 100644 index 00000000..ace8b1a2 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/limited_partner.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=limited_partner. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const LimitedPartnerPlugin = makeGenericPlugin("limited_partner"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js b/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js new file mode 100644 index 00000000..39706ce5 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/policy_advisor.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=policy_advisor. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const PolicyAdvisorPlugin = makeGenericPlugin("policy_advisor"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/regulator.js b/apps/worker/test-dist-q/services/personas/kinds/regulator.js new file mode 100644 index 00000000..0219b79f --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/regulator.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=regulator. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const RegulatorPlugin = makeGenericPlugin("regulator"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/service_provider.js b/apps/worker/test-dist-q/services/personas/kinds/service_provider.js new file mode 100644 index 00000000..4abb6751 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/service_provider.js @@ -0,0 +1,7 @@ +// Task #3: thin plugin wrapper for kind=service_provider. Delegates to the +// generic plugin which drives candidate selection from taxonomy +// roles+targets and scoring via the person-graph scorer for person +// targets. Kept as a discrete file so the plugin-per-kind contract +// is satisfied and future per-kind customization has a home. +import { makeGenericPlugin } from "./_generic"; +export const ServiceProviderPlugin = makeGenericPlugin("service_provider"); diff --git a/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js b/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js new file mode 100644 index 00000000..981f0019 --- /dev/null +++ b/apps/worker/test-dist-q/services/personas/kinds/taxonomy.js @@ -0,0 +1,86 @@ +// Task #3: Expand persona kinds taxonomy. +// +// Single source of truth for the persona-kind taxonomy. Both the form +// (consumed via GET /api/personas/taxonomy) and the matcher dispatcher +// read from this module. No taxonomy duplication in HTML or in plugin +// code — plugins reference KINDS[kind] to discover their group, label, +// allowed criteria sections, and required hint fields. +// Hint-field metadata for the form (label, type, options). +export const HINTS = { + subtype: { label: "Subtype", type: "select", options: ["lawyer", "banker", "operator", "politician", "scout", "advisor", "board_member"] }, + aum_band: { label: "AUM band", type: "select", options: ["<$50M", "$50M-$250M", "$250M-$1B", "$1B-$5B", ">$5B"] }, + stage_focus: { label: "Stage focus", type: "text", placeholder: "pre_seed, seed, series_a" }, + founded_count: { label: "Companies founded (min)", type: "number" }, + prior_exits: { label: "Prior exits (min)", type: "number" }, + domain_expertise: { label: "Domain expertise", type: "text", placeholder: "fintech, dev_tools" }, +}; +const COMMON = ["geography", "industry", "signals", "tuning"]; +const PERSON_BASE = [...COMMON, "buyer_profile"]; +const COMPANY_BASE = ["sizing", ...COMMON, "tech_stack"]; +export const KINDS_LIST = [ + // ---- Sales + { kind: "account_company", group: "Sales", label: "Account (company)", sections: COMPANY_BASE, hints: [], targets: "company", roles: [] }, + { kind: "buyer_person", group: "Sales", label: "Buyer (person)", sections: PERSON_BASE, hints: [], targets: "person", roles: ["buyer", "decision_maker", "champion"] }, + // ---- Capital + { kind: "investor_firm", group: "Capital", label: "Investor firm", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band", "stage_focus"], targets: "fund", roles: ["investor_firm"] }, + { kind: "investor_person", group: "Capital", label: "Investor (person)", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["stage_focus"], targets: "person", roles: ["investor", "vc", "gp", "partner_at_firm"] }, + { kind: "angel_individual", group: "Capital", label: "Angel investor", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["angel", "investor"] }, + { kind: "limited_partner", group: "Capital", label: "Limited partner", sections: ["geography", "signals", "tuning"], hints: ["aum_band"], targets: "person", roles: ["limited_partner", "lp"] }, + { kind: "venture_partner", group: "Capital", label: "Venture partner", sections: ["geography", "industry", "signals", "buyer_profile", "tuning"], hints: ["subtype", "domain_expertise"], targets: "person", roles: ["venture_partner", "advisor", "scout"] }, + // ---- People + { kind: "founder", group: "People", label: "Founder", sections: ["geography", "industry", "signals", "tuning"], hints: ["founded_count", "prior_exits", "domain_expertise"], targets: "person", roles: ["founder", "ceo"] }, + { kind: "co_founder_match", group: "People", label: "Co-founder match", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["founder", "engineer", "designer"] }, + { kind: "executive_hire", group: "People", label: "Executive hire", sections: PERSON_BASE, hints: ["domain_expertise"], targets: "person", roles: ["executive", "vp", "c_suite"] }, + { kind: "engineering_hire", group: "People", label: "Engineering hire", sections: ["geography", "tech_stack", "signals", "buyer_profile", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["engineer", "ic"] }, + { kind: "fractional_executive", group: "People", label: "Fractional executive", sections: PERSON_BASE, hints: ["domain_expertise", "prior_exits"], targets: "person", roles: ["fractional", "advisor", "executive"] }, + // ---- Partnerships + { kind: "channel_partner", group: "Partnerships", label: "Channel partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["partner", "reseller"] }, + { kind: "integration_partner", group: "Partnerships", label: "Integration partner", sections: ["sizing", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["partner", "integration"] }, + { kind: "design_partner", group: "Partnerships", label: "Design partner", sections: COMPANY_BASE, hints: [], targets: "company", roles: ["customer", "prospect"] }, + { kind: "beta_tester", group: "Partnerships", label: "Beta tester", sections: ["geography", "industry", "tech_stack", "signals", "tuning"], hints: [], targets: "company", roles: ["customer", "prospect", "beta"] }, + // ---- Influence (no tech_stack — content/coverage focused) + { kind: "journalist_analyst", group: "Influence", label: "Journalist / analyst", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["journalist", "analyst", "press"] }, + { kind: "thought_leader", group: "Influence", label: "Thought leader", sections: ["geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["influencer", "thought_leader", "speaker"] }, + { kind: "academic_researcher", group: "Influence", label: "Academic researcher", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["researcher", "academic", "professor"] }, + // ---- Public Sector + { kind: "government_grant_officer", group: "Public Sector", label: "Government grant officer", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["government", "grant_officer", "program_officer"] }, + { kind: "regulator", group: "Public Sector", label: "Regulator", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["regulator", "agency_official"] }, + { kind: "policy_advisor", group: "Public Sector", label: "Policy advisor", sections: ["geography", "industry", "tuning"], hints: ["domain_expertise"], targets: "person", roles: ["policy_advisor", "staffer", "aide"] }, + // ---- Operational + { kind: "service_provider", group: "Operational", label: "Service provider", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["domain_expertise"], targets: "company", roles: ["vendor", "service_provider", "agency"] }, + { kind: "acquirer", group: "Operational", label: "Acquirer", sections: ["sizing", "geography", "industry", "signals", "tuning"], hints: ["aum_band"], targets: "company", roles: ["acquirer", "strategic"] }, + { kind: "competitor", group: "Operational", label: "Competitor", sections: ["sizing", "geography", "industry", "tech_stack", "tuning"], hints: [], targets: "company", roles: ["competitor"] }, +]; +export const KINDS = Object.fromEntries(KINDS_LIST.map((k) => [k.kind, k])); +export const ALL_KIND_KEYS = KINDS_LIST.map((k) => k.kind); +// Legacy values from before Task #3 — map to the closest new kind so +// existing personas keep working without a backfill migration. +export const LEGACY_KIND_MAP = { + account: "account_company", + buyer: "buyer_person", +}; +export function resolveKind(raw) { + if (!raw) + return null; + if (KINDS[raw]) + return raw; + if (LEGACY_KIND_MAP[raw]) + return LEGACY_KIND_MAP[raw]; + return null; +} +export function isValidKind(raw) { + return resolveKind(raw) !== null; +} +// Grouped view for the form's grouped