From b14d2fd0c7c6cda42c72a61d3b104f0beeefc3b3 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 11 Aug 2026 15:10:58 +0000 Subject: [PATCH 1/3] feat(ai): the outreach composer, and the checks that make it trustworthy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds `packages/ai` — the only package in the tree that talks to a model. The composer writes the message body for a recommendation, and everything around it exists to make that output safe to show a reviewer. Two properties are structural rather than requested: 1. It can only see grounded inputs. The prompt is assembled from stored evidence, stored facts and the customer's own offering. There is no path by which the model learns something it may not cite. 2. Its output is checked, not trusted. Eight deterministic §14.2 gates run on every draft — grounding overlap, unsupported claims, identity confidence, flattery, spam patterns, cross-prospect duplication, sensitive topics, policy. A draft that fails is rewritten once, naming the exact invented fragments; still failing, it is withheld. Withheld is a working state, not an error. The card keeps the prospect, the evidence and the recommended action, and the reviewer writes the message. A bad draft next to a caveat is still a bad draft someone might approve. The undecidable question "is this claim true?" is inverted into a decidable one: does every specific assertion — quoted phrase, number, mid-sentence proper noun — appear in stored evidence? That is what `checks.ts` answers, and it is why no model output reaches a human unexamined. Also: - `POST /recommendations/:id/draft` composes or recomposes on demand, returns 200 with `drafted: false` and a reason when withholding, and audits it. A user-edited draft is never discarded by a recompose. - The approval card gains "Write draft" / "Rewrite", and states plainly why no draft exists rather than offering a retry that cannot help. - Prompt caching splits the stable offering/voice prefix from the per-prospect half, so a campaign pays for the prefix once. - The composer is optional infrastructure: without ANTHROPIC_API_KEY the queue still runs end to end and drafting returns 503. Refusing to boot would take the system down for a feature designed to produce nothing. No model decides a merge, a score, or a policy outcome. 367 tests pass. Co-Authored-By: Claude Opus 5 (1M context) --- .env.example | 6 + README.md | 18 +- apps/api/package.json | 3 +- apps/api/src/app.test.ts | 101 ++++++ apps/api/src/app.ts | 66 ++++ apps/server/package.json | 3 +- apps/server/src/index.ts | 19 ++ apps/web/components/approval-card.tsx | 128 +++++-- apps/worker/package.json | 3 +- apps/worker/src/pipeline.ts | 19 ++ bun.lock | 22 ++ docs/prd-implementation-map.md | 7 +- packages/ai/package.json | 18 + packages/ai/src/checks.test.ts | 307 +++++++++++++++++ packages/ai/src/checks.ts | 465 ++++++++++++++++++++++++++ packages/ai/src/composer.test.ts | 292 ++++++++++++++++ packages/ai/src/composer.ts | 297 ++++++++++++++++ packages/ai/src/draft.test.ts | 184 ++++++++++ packages/ai/src/draft.ts | 239 +++++++++++++ packages/ai/src/index.ts | 41 +++ packages/ai/src/model.ts | 143 ++++++++ tsconfig.json | 4 +- 22 files changed, 2353 insertions(+), 32 deletions(-) create mode 100644 packages/ai/package.json create mode 100644 packages/ai/src/checks.test.ts create mode 100644 packages/ai/src/checks.ts create mode 100644 packages/ai/src/composer.test.ts create mode 100644 packages/ai/src/composer.ts create mode 100644 packages/ai/src/draft.test.ts create mode 100644 packages/ai/src/draft.ts create mode 100644 packages/ai/src/index.ts create mode 100644 packages/ai/src/model.ts diff --git a/.env.example b/.env.example index d89deb7..cb06652 100644 --- a/.env.example +++ b/.env.example @@ -39,7 +39,13 @@ X_API_SECRET= # ---------------------------------------------------------------------- LLM # Never exposed to the browser, and never included in a model prompt alongside # customer OAuth tokens (PRD §34). +# +# Optional. Unset, the composer is disabled: signals, resolution, scoring and +# the approval queue all still run, and the reviewer writes the message. It is +# never consulted for a policy, identity or scoring decision (PRD §1.1.8). ANTHROPIC_API_KEY= +# Overrides the default model (claude-opus-5). +ANTHROPIC_MODEL= # -------------------------------------------------------------------- queue # Optional until queued jobs need durability. diff --git a/README.md b/README.md index abb2504..10e5fb7 100644 --- a/README.md +++ b/README.md @@ -9,10 +9,10 @@ platform, the prospect, or your own rate limits say it should not. ## Status -Foundation. The deterministic core, the API, the PWA and one live provider are -built and tested; a prospect can go from a GitHub handle to a card in the -approval queue today. The LLM layer — the outreach composer and the rest of the -§20 agent suite — is not built. See +Foundation. The deterministic core, the API, the PWA, one live provider and the +outreach composer are built and tested; a prospect can go from a GitHub handle +to a drafted, checked card in the approval queue today. The rest of the §20 +agent suite is not built. See [`docs/prd-implementation-map.md`](docs/prd-implementation-map.md) for exactly what exists. @@ -43,6 +43,7 @@ apps/ web/ Next.js 16 mobile-first PWA worker/ background jobs: signal expiry, rescoring, privacy work packages/ + ai/ the only package that talks to a model: composer + quality gates domain/ canonical types — depends on nothing db/ Turso/libSQL client and migration runner policy/ the deterministic policy engine @@ -88,6 +89,15 @@ state. Suppression, a flipped feature flag, a spent rate limit, or a dropped identity confidence all block an approval that would have been fine yesterday, and the response names the gate that stopped it. +**A draft that fails its checks is withheld, not shown with a warning.** The +composer only ever sees stored evidence, stored facts and your own offering — +there is no path by which it learns something it may not cite. Its output then +runs deterministic gates: every specific assertion must appear in that +evidence, or the draft is rejected and rewritten once naming the exact invented +fragments. Still failing, nothing is shown. The card keeps the prospect, the +evidence and the recommended action, and you write the message. A bad draft +next to a caveat is still a bad draft someone might approve. + **Evidence combines with noisy-OR, so weak signals never reach certainty.** Identity resolution and intent scoring both use `1 - Π(1 - eᵢ)`. Two 0.5 observations give 0.75, not 1.0. "Same name, same city" — the classic diff --git a/apps/api/package.json b/apps/api/package.json index 340eb8c..61e4dff 100644 --- a/apps/api/package.json +++ b/apps/api/package.json @@ -19,6 +19,7 @@ "@outreachgraph/scoring": "workspace:*", "@outreachgraph/signals": "workspace:*", "hono": "^4.6.14", - "zod": "^3.24.1" + "zod": "^3.24.1", + "@outreachgraph/ai": "workspace:*" } } diff --git a/apps/api/src/app.test.ts b/apps/api/src/app.test.ts index 6090eab..6f0c21d 100644 --- a/apps/api/src/app.test.ts +++ b/apps/api/src/app.test.ts @@ -1,5 +1,6 @@ import { afterEach, describe, expect, test } from 'bun:test'; import type { Hono } from 'hono'; +import { StubModel } from '@outreachgraph/ai'; import { createApp } from './app'; import type { AppEnv, RequestActor } from './context'; import { seedDatabase, SEED, type SeededDatabase } from './test-seed'; @@ -484,3 +485,103 @@ describe('errors', () => { expect(response.status).toBe(400); }); }); + +describe('drafting on demand (PRD §14)', () => { + /** The seed's evidence is a cross-border payouts complaint. */ + const GROUNDED = 'Settlement taking days on cross-border payouts was our problem too.'; + + async function withModel( + label: string, + responses: string | readonly string[], + ): Promise> { + const seeded = await seedDatabase(label); + active = seeded; + + return createApp({ + db: seeded.db, + authenticate: async () => ACTOR, + model: new StubModel(responses), + }); + } + + test('returns 503 when no model is configured', async () => { + const { app } = await harness('draft-nomodel'); + const response = await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + expect(response.status).toBe(503); + expect((await response.json()).error.code).toBe('composer_unavailable'); + }); + + test('composes a grounded message and returns it', async () => { + const app = await withModel('draft-ok', GROUNDED); + const response = await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + expect(response.status).toBe(200); + const body = await response.json(); + expect(body.drafted).toBe(true); + expect(body.body).toBe(GROUNDED); + }); + + test('replaces the previous draft rather than stacking a second one', async () => { + const app = await withModel('draft-replace', GROUNDED); + await post(app, `/recommendations/${SEED.recommendationId}/draft`); + await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + const rows = await active!.db.execute({ + sql: 'SELECT count(*) AS n FROM drafts WHERE recommendation_id = ?', + args: [SEED.recommendationId], + }); + expect(Number(rows.rows[0]?.n)).toBe(1); + }); + + test('reports a withheld draft instead of surfacing an invented one', async () => { + const app = await withModel('draft-withheld', [ + 'On cross-border payouts, Fluxwire solved this for us.', + 'Your payouts note — we moved to Fluxwire.', + ]); + + const response = await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + // Not an error: refusing to write is a normal, expected answer. + expect(response.status).toBe(200); + const body = await response.json(); + expect(body.drafted).toBe(false); + expect(body.reason).toBe('failed_checks'); + expect(body.unsupported).toContain('Fluxwire'); + expect(body.body).toBeUndefined(); + }); + + test('a withheld draft is audited', async () => { + const app = await withModel('draft-audit', [ + 'On cross-border payouts, Fluxwire solved this for us.', + 'Your payouts note — we moved to Fluxwire.', + ]); + await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + const rows = await active!.db.execute( + "SELECT count(*) AS n FROM audit_events WHERE event_type = 'draft.withheld'", + ); + expect(Number(rows.rows[0]?.n)).toBe(1); + }); + + test('an unknown recommendation is a 404', async () => { + const app = await withModel('draft-404', GROUNDED); + expect((await post(app, '/recommendations/rec_missing/draft')).status).toBe(404); + }); + + test('a draft the user edited is never discarded by a recompose', async () => { + const app = await withModel('draft-edited', GROUNDED); + await active!.db.execute({ + sql: 'UPDATE drafts SET edited_by_user = 1 WHERE id = ?', + args: [SEED.draftId], + }); + + await post(app, `/recommendations/${SEED.recommendationId}/draft`); + + const rows = await active!.db.execute({ + sql: 'SELECT edited_by_user FROM drafts WHERE id = ?', + args: [SEED.draftId], + }); + expect(rows.rows).toHaveLength(1); + }); +}); diff --git a/apps/api/src/app.ts b/apps/api/src/app.ts index c2e06f0..dff5e6e 100644 --- a/apps/api/src/app.ts +++ b/apps/api/src/app.ts @@ -33,6 +33,7 @@ import { workspacesForUser, } from './auth'; import { evaluatePolicy, isExecutable, POLICY_VERSION } from '@outreachgraph/policy'; +import { draftForRecommendation, type TextModel } from '@outreachgraph/ai'; import { ApiError, canApprove, type AppEnv, type RequestActor } from './context'; import * as repo from './repository'; @@ -50,6 +51,11 @@ export interface AppOptions { readonly serviceToken?: string; /** Set false for plain-HTTP local development so the cookie still sets. */ readonly secureCookies?: boolean; + /** + * Writes outreach drafts. Omit to run without a language model — every + * other route works unchanged and drafting returns 503. + */ + readonly model?: TextModel; readonly version?: string; readonly commitHash?: string; } @@ -451,6 +457,66 @@ export function createApp(options: AppOptions): Hono { }); }); + /** + * Compose (or recompose) the message for a recommendation. + * + * Separate from generation because a draft is optional: the pipeline places + * the card in the queue whether or not the composer produced anything, and + * the reviewer can ask for one here. A refusal to write is a normal answer, + * returned with the reason and the specific fragments that failed grounding + * — the alternative is showing an invented message, which is worse. + */ + api.post('/recommendations/:id/draft', async (c) => { + const actor = c.get('actor'); + const db = c.get('db'); + + if (!options.model) { + throw new ApiError( + 503, + 'composer_unavailable', + 'no language model is configured; set ANTHROPIC_API_KEY to enable drafting', + ); + } + + const recommendation = await repo.getRecommendation(db, actor.workspaceId, c.req.param('id')); + if (!recommendation) throw ApiError.notFound('recommendation'); + + // Recomposing replaces the previous draft; the writer asked for a rewrite. + await db.execute({ + sql: 'DELETE FROM drafts WHERE recommendation_id = ? AND edited_by_user = 0', + args: [recommendation.id], + }); + + const result = await draftForRecommendation(db, options.model, recommendation.id); + + if (!result.ok) { + await repo.audit(db, { + workspaceId: actor.workspaceId, + actorKind: 'user', + actorId: actor.userId, + eventType: 'draft.withheld', + entityKind: 'recommendation', + entityId: recommendation.id, + detail: { reason: result.reason, unsupported: result.unsupported ?? [] }, + }); + + return c.json( + { + drafted: false, + reason: result.reason, + ...(result.unsupported ? { unsupported: result.unsupported } : {}), + }, + 200, + ); + } + + const draft = await queryOne<{ body: string }>(db, 'SELECT body FROM drafts WHERE id = ?', [ + result.draftId!, + ]); + + return c.json({ drafted: true, draftId: result.draftId, body: draft?.body }); + }); + api.post('/recommendations/:id/skip', async (c) => { const actor = c.get('actor'); const db = c.get('db'); diff --git a/apps/server/package.json b/apps/server/package.json index d2ba3fd..48d15bc 100644 --- a/apps/server/package.json +++ b/apps/server/package.json @@ -19,6 +19,7 @@ "@outreachgraph/scoring": "workspace:*", "@outreachgraph/signals": "workspace:*", "hono": "^4.6.14", - "zod": "^3.24.1" + "zod": "^3.24.1", + "@outreachgraph/ai": "workspace:*" } } diff --git a/apps/server/src/index.ts b/apps/server/src/index.ts index 69687d3..8dc38ec 100644 --- a/apps/server/src/index.ts +++ b/apps/server/src/index.ts @@ -20,6 +20,7 @@ */ import { closeDatabase, getDatabase, migrate, queryAll } from '@outreachgraph/db'; +import { ClaudeModel } from '@outreachgraph/ai'; import { createApp } from '../../api/src/app'; import { pruneSessions } from '../../api/src/auth'; import { expireSignals, processDeletion } from '../../worker/src/jobs'; @@ -60,9 +61,27 @@ if (process.env.RUN_MIGRATIONS !== 'false') { } } +// -------------------------------------------------------------------- model +/** + * The composer is optional infrastructure. + * + * Without a key the product still works: signals are ingested, prospects + * resolved, recommendations scored and queued — the reviewer writes the + * message. Refusing to boot over a missing key would take the whole system + * down for a feature that is, by design, allowed to produce nothing. + */ +const model = process.env.ANTHROPIC_API_KEY + ? new ClaudeModel({ + ...(process.env.ANTHROPIC_MODEL ? { model: process.env.ANTHROPIC_MODEL } : {}), + }) + : undefined; + +if (!model) console.log('no ANTHROPIC_API_KEY: drafting disabled, queue still runs'); + // ---------------------------------------------------------------------- api const api = createApp({ db, + ...(model ? { model } : {}), ...(process.env.API_TOKEN ? { serviceToken: process.env.API_TOKEN } : {}), // Cookies must not be Secure over plain HTTP, or local development can // never hold a session. diff --git a/apps/web/components/approval-card.tsx b/apps/web/components/approval-card.tsx index 61f0006..bdf6190 100644 --- a/apps/web/components/approval-card.tsx +++ b/apps/web/components/approval-card.tsx @@ -18,13 +18,84 @@ import type { ApprovalCard as Card } from '../lib/types'; * policy gate that stopped it, and that reason is shown verbatim. Silently * dropping it would leave the user thinking they had sent something. */ +/** + * Why no draft was written, in the reviewer's language. + * + * The reason is shown rather than hidden behind "try again", because every + * one of these is a fact about the evidence: retrying will not change it, and + * the honest next step is for the reviewer to write the message themselves. + */ +function explainWithholding(reason: string, unsupported?: string[]): string { + switch (reason) { + case 'no_evidence': + return 'No draft: nothing quotable was captured for this signal, and a personalised message with nothing behind it is worse than none.'; + case 'no_trigger_signal': + return 'No draft: this recommendation has no signal to reference.'; + case 'failed_checks': + return unsupported?.length + ? `No draft: the wording kept asserting things nothing supports — ${unsupported.join(', ')}.` + : 'No draft: the wording did not pass the quality checks.'; + case 'model_refused': + case 'empty': + return 'No draft: the writer declined to produce one.'; + default: + return `No draft (${reason}).`; + } +} + export function ApprovalCard({ card }: { card: Card }) { const router = useRouter(); const [busy, setBusy] = useState(); const [error, setError] = useState(); const [body, setBody] = useState(card.draft_body ?? ''); + // What the composer produced, so an approval can tell an edit from the + // original rather than reporting every composed message as user-written. + const [original, setOriginal] = useState(card.draft_body ?? ''); + const [withheld, setWithheld] = useState(); const [editing, setEditing] = useState(false); + /** + * Ask the composer for wording. + * + * A withheld draft comes back 200 with a reason — it is the designed + * outcome when nothing in the evidence supports a message, not an error. + * It is shown as a note so the reviewer knows to write it themselves. + */ + async function compose() { + setBusy('draft'); + setError(undefined); + setWithheld(undefined); + + try { + const response = await fetch(`/api/v1/recommendations/${card.id}/draft`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + credentials: 'same-origin', + body: '{}', + }); + + const payload = await response.json().catch(() => ({})); + + if (!response.ok) { + setError(payload?.error?.message ?? `that failed (${response.status})`); + return; + } + + if (!payload.drafted) { + setWithheld(explainWithholding(payload.reason, payload.unsupported)); + return; + } + + setBody(payload.body); + setOriginal(payload.body); + router.refresh(); + } catch { + setError('could not reach the server'); + } finally { + setBusy(undefined); + } + } + async function act(action: 'approve' | 'skip', payload?: Record) { setBusy(action); setError(undefined); @@ -108,25 +179,44 @@ export function ApprovalCard({ card }: { card: Card }) {

- {card.draft_body ? ( -
+
+

Draft

- {editing ? ( -