diff --git a/examples/with-typesafe/.gitignore b/examples/with-typesafe/.gitignore new file mode 100644 index 0000000..f650315 --- /dev/null +++ b/examples/with-typesafe/.gitignore @@ -0,0 +1,27 @@ +# See https://help.github.com/articles/ignoring-files/ for more about ignoring files. + +# dependencies +/node_modules + +# next.js +/.next/ +/out/ + +# production +/build + +# debug +npm-debug.log* +yarn-debug.log* +yarn-error.log* +.pnpm-debug.log* + +# env files +.env* + +# vercel +.vercel + +# typescript +*.tsbuildinfo +next-env.d.ts \ No newline at end of file diff --git a/examples/with-typesafe/README.md b/examples/with-typesafe/README.md new file mode 100644 index 0000000..3b9dee7 --- /dev/null +++ b/examples/with-typesafe/README.md @@ -0,0 +1,26 @@ +# JEV-based PDF form filling (Typesafe × SimplePDF) + +_Built with Next.js, Tailwind CSS, `@simplepdf/react-embed-pdf`, and Typesafe's **JEV** ("System One") model._ + +Pick a form (an **IRS W-9** or a **medication prior-authorization**), and JEV auto-fills it from a demo record in the SimplePDF editor: it maps each CSV column to a field, ticks the right checkboxes and picks constrained-field options, and gates every write on a calibrated confidence. Fills below the auto-threshold surface as **low-confidence** items to confirm, a JEV plausibility pass flags **fictional / implausible** values, and any field it can't fill is a manual-entry prompt. A human confirms or fixes each one, re-validates, and finalizes. + +The shape is the message: **code owns control flow, JEV answers narrow structured questions** (which column fits this field? should this box be checked? is this value plausible?). No agent, no hallucination. + +## Run + +```sh +npm install +npm run dev +``` + +Open `http://localhost:3001`. The editor loads under the whitelisted `spdf-jev` demo origin (no Pro account needed). + +## JEV API key (BYOK, in-memory) + +The JEV key is **yours** and is held **in memory only**: no server, no persistence, no localStorage. The call is proxied same-origin through `/api/jev` (JEV has no browser CORS), and the key is forwarded, never stored. Get a key at `https://console.typesafe.ai/settings/keys`. + +**What reaches JEV:** field labels, the demo CSV record, and the filled text values (for the plausibility pass). Checkbox/option states, signature data, and picture data are never sent. + +For local dev you can pre-seed the key: copy it into `.env.local` as `NEXT_PUBLIC_TYPESAFE_API_KEY=` (gitignored, never committed). The shipped path is still typing the key into the UI. + +> Illustrative demo, not tax advice. diff --git a/examples/with-typesafe/app/api/jev/[...path]/route.ts b/examples/with-typesafe/app/api/jev/[...path]/route.ts new file mode 100644 index 0000000..8671cfc --- /dev/null +++ b/examples/with-typesafe/app/api/jev/[...path]/route.ts @@ -0,0 +1,46 @@ +// Same-origin proxy to JEV. api.typesafe.ai sends no Access-Control-Allow-Origin, +// so a browser-direct call is blocked; this route forwards it server-side (no CORS) +// with the caller's BYOK Authorization header, which it never stores. The path is +// allowlisted to the endpoints the SDK uses, so it is not an open relay. +// CF: plans/P107-typesafe-jev-example.md + +const JEV_UPSTREAM = "https://api.typesafe.ai" +const ALLOWED_PATHS = new Set(["v1/systemone"]) +const UPSTREAM_TIMEOUT_MS = 15_000 + +export const POST = async ( + request: Request, + context: { params: Promise<{ path: string[] }> }, +): Promise => { + const { path } = await context.params + const joined = path.join("/") + if (!ALLOWED_PATHS.has(joined)) { + return Response.json({ error: "Not a permitted JEV path" }, { status: 404 }) + } + + const authorization = request.headers.get("authorization") + const body = await request.text() + + try { + const upstream = await fetch(`${JEV_UPSTREAM}/${joined}`, { + method: "POST", + headers: { + "content-type": "application/json", + ...(authorization !== null ? { authorization } : {}), + }, + body, + // Never follow a redirect: a 3xx could otherwise replay this POST (and the BYOK + // Authorization header) to an arbitrary host. The upstream is a fixed API origin. + redirect: "error", + // Forward the caller's cancellation AND cap the round trip. + signal: AbortSignal.any([request.signal, AbortSignal.timeout(UPSTREAM_TIMEOUT_MS)]), + }) + const responseBody = await upstream.text() + return new Response(responseBody, { + status: upstream.status, + headers: { "content-type": upstream.headers.get("content-type") ?? "application/json" }, + }) + } catch { + return Response.json({ error: "JEV upstream is unavailable" }, { status: 502 }) + } +} diff --git a/examples/with-typesafe/app/globals.css b/examples/with-typesafe/app/globals.css new file mode 100644 index 0000000..e448a8e --- /dev/null +++ b/examples/with-typesafe/app/globals.css @@ -0,0 +1,75 @@ +@import "tailwindcss"; + +/* TypeSafe-inspired dark palette (typesafe.ai): near-black canvas, pink/magenta + brand accents, calibrated-confidence semantics (green=valid, amber=low, red=error). + Tokens are defined once on :root and mapped into Tailwind v4's theme below. */ +@theme inline { + --color-background: hsl(var(--background)); + --color-foreground: hsl(var(--foreground)); + --color-card: hsl(var(--card)); + --color-card-foreground: hsl(var(--card-foreground)); + --color-popover: hsl(var(--popover)); + --color-popover-foreground: hsl(var(--popover-foreground)); + --color-primary: hsl(var(--primary)); + --color-primary-foreground: hsl(var(--primary-foreground)); + --color-secondary: hsl(var(--secondary)); + --color-secondary-foreground: hsl(var(--secondary-foreground)); + --color-muted: hsl(var(--muted)); + --color-muted-foreground: hsl(var(--muted-foreground)); + --color-accent: hsl(var(--accent)); + --color-accent-foreground: hsl(var(--accent-foreground)); + --color-destructive: hsl(var(--destructive)); + --color-destructive-foreground: hsl(var(--destructive-foreground)); + --color-border: hsl(var(--border)); + --color-input: hsl(var(--input)); + --color-ring: hsl(var(--ring)); + --color-brand: hsl(var(--brand)); + --color-brand-2: hsl(var(--brand-2)); + --color-valid: hsl(var(--valid)); + --color-warn: hsl(var(--warn)); + --color-error: hsl(var(--error)); + --color-info: hsl(var(--info)); +} + +:root { + --background: 0 0% 12%; + --foreground: 0 0% 98%; + --card: 0 0% 15%; + --card-foreground: 0 0% 98%; + --popover: 0 0% 12%; + --popover-foreground: 0 0% 98%; + --primary: 345 82% 74%; + --primary-foreground: 0 0% 12%; + --secondary: 0 0% 18%; + --secondary-foreground: 0 0% 98%; + --muted: 0 0% 18%; + --muted-foreground: 0 0% 52%; + --accent: 0 0% 18%; + --accent-foreground: 0 0% 98%; + --destructive: 0 84% 60%; + --destructive-foreground: 0 0% 98%; + --border: 0 0% 22%; + --input: 0 0% 22%; + --ring: 345 82% 74%; + --brand: 345 82% 74%; + --brand-2: 314 58% 59%; + --valid: 152 95% 34%; + --warn: 38 92% 50%; + --error: 0 84% 60%; + --info: 175 90% 36%; +} + +@layer base { + * { + @apply border-border; + } + body { + @apply bg-background text-foreground; + font-family: ui-sans-serif, system-ui, -apple-system, "Segoe UI", Roboto, Helvetica, Arial, sans-serif; + } + /* Tailwind v4 drops the default button pointer; restore it for interactive elements. */ + button:not(:disabled), + [role="button"]:not([aria-disabled="true"]) { + cursor: pointer; + } +} diff --git a/examples/with-typesafe/app/layout.tsx b/examples/with-typesafe/app/layout.tsx new file mode 100644 index 0000000..66feefd --- /dev/null +++ b/examples/with-typesafe/app/layout.tsx @@ -0,0 +1,21 @@ +import type { Metadata } from 'next' +import type { ReactElement, ReactNode } from 'react' +import './globals.css' + +export const metadata: Metadata = { + title: 'JEV-based PDF form filling · SimplePDF', + description: + 'Fill a PDF from a CSV with JEV (Typesafe): auto-fill the fields, raise low-confidence and validation issues, human signs off. Built on the SimplePDF editor.', +} + +export default function RootLayout({ + children, +}: Readonly<{ + children: ReactNode +}>): ReactElement { + return ( + + {children} + + ) +} diff --git a/examples/with-typesafe/app/loading.tsx b/examples/with-typesafe/app/loading.tsx new file mode 100644 index 0000000..f15322a --- /dev/null +++ b/examples/with-typesafe/app/loading.tsx @@ -0,0 +1,3 @@ +export default function Loading() { + return null +} diff --git a/examples/with-typesafe/app/page.tsx b/examples/with-typesafe/app/page.tsx new file mode 100644 index 0000000..0e4e52c Binary files /dev/null and b/examples/with-typesafe/app/page.tsx differ diff --git a/examples/with-typesafe/benchmark/backends.ts b/examples/with-typesafe/benchmark/backends.ts new file mode 100644 index 0000000..be2c44c --- /dev/null +++ b/examples/with-typesafe/benchmark/backends.ts @@ -0,0 +1,147 @@ +// Benchmark-only (delete with benchmark/ before the PR). Two backends run the SAME work and return +// BOTH their timing AND their decisions, so the harness can check the outputs are equivalent before +// trusting the latency numbers. JEV via the Typesafe SDK (structured choice/noul); any +// OpenAI-compatible chat model via JSON mode. Node runs server-side, so no CORS and no proxy. +import { TypeSafeClient, choice, noul } from "@typesafe-ai/sdk" +import type { BenchField, BenchInput } from "./fixtures" + +export type PhaseTiming = { mapMs: number; classifyMs: number } +export type ModelOutput = { + timing: PhaseTiming + mapping: Record // fieldId -> chosen column / option / "none" + plausibility: Record // field label -> genuine score (0..1) +} +// `extraBody` is merged into each request, so a reasoning model can be tuned per provider without a +// code change (e.g. { reasoning_effort: "low" }, { thinking: false }). +export type OpenAICompatibleConfig = { baseUrl: string; model: string; apiKey: string; extraBody: Record } + +const JEV_BASE_URL = "https://api.typesafe.ai" +const NONE = "none" +const REQUEST_TIMEOUT_MS = 60_000 + +const isRecord = (value: unknown): value is Record => typeof value === "object" && value !== null + +const isFillable = (field: BenchField): boolean => field.type !== "SIGNATURE" && field.type !== "PICTURE" +const fieldOptions = (field: BenchField): string[] | null => + field.options !== null && field.options.length > 0 ? field.options : null + +// --- JEV (Typesafe SDK) — the exact structured questions the app asks ------------------------ + +const jevMappingQuestions = (input: BenchInput): Record> => { + const columnCriteria: Record = { + ...Object.fromEntries(input.record.columns.map((column) => [column, `CSV column "${column}" = "${input.record.values[column] ?? ""}"`])), + [NONE]: "No CSV column fits this field", + } + return Object.fromEntries( + input.fields + .filter(isFillable) + .map((field): [string, ReturnType] => { + const options = fieldOptions(field) + if (options !== null) { + return [ + field.fieldId, + choice(`Given the record, which value fits the field labeled "${field.name}" (type ${field.type})? Answer "${NONE}" to leave it blank.`, { + ...Object.fromEntries(options.map((option) => [option, `The correct value for this field is "${option}"`])), + [NONE]: "No option fits the record; leave the field blank", + }), + ] + } + return [ + field.fieldId, + choice(`Which CSV column should fill the form field labeled "${field.name}" (type ${field.type})? Answer "${NONE}" if no column fits.`, columnCriteria), + ] + }), + ) +} + +const jevPlausibilityQuestions = (input: BenchInput): Record> => + Object.fromEntries( + input.filled.map((field): [string, ReturnType] => [ + field.name, + noul(`Is "${field.value}" a genuine, plausible value for the field labeled "${field.name}" (not fictional, a placeholder, or obviously wrong)?`), + ]), + ) + +export const runJev = async (apiKey: string, input: BenchInput): Promise => { + const client = new TypeSafeClient({ apiKey, baseURL: JEV_BASE_URL }) + + const mapStart = performance.now() + const { answers: mapAnswers } = await client.systemOne({ state: input.record.values, questions: jevMappingQuestions(input) }) + const mapMs = performance.now() - mapStart + const mapping = Object.fromEntries(Object.entries(mapAnswers).map(([fieldId, answer]) => [fieldId, answer.choice])) + + const plausibilityState = { fields: Object.fromEntries(input.filled.map((field) => [field.name, field.value])) } + const classifyStart = performance.now() + const { answers: plausAnswers } = await client.systemOne({ state: plausibilityState, questions: jevPlausibilityQuestions(input) }) + const classifyMs = performance.now() - classifyStart + const plausibility = Object.fromEntries(Object.entries(plausAnswers).map(([label, answer]) => [label, answer.noul])) + + return { timing: { mapMs, classifyMs }, mapping, plausibility } +} + +// --- OpenAI-compatible chat model (JSON mode) — the same work, one batched call per phase ------- + +const chatJson = async (config: OpenAICompatibleConfig, system: string, user: string): Promise> => { + const response = await fetch(`${config.baseUrl.replace(/\/$/, "")}/chat/completions`, { + method: "POST", + headers: { "content-type": "application/json", authorization: `Bearer ${config.apiKey}` }, + body: JSON.stringify({ + model: config.model, + messages: [ + { role: "system", content: system }, + { role: "user", content: user }, + ], + response_format: { type: "json_object" }, + temperature: 0, + ...config.extraBody, + }), + // Fail fast instead of hanging the whole benchmark if the endpoint stalls. + signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS), + }) + if (!response.ok) { + const detail = await response.text() + throw new Error(`Model request failed (${response.status}): ${detail.slice(0, 300)}`) + } + const data: unknown = await response.json() + const content = isRecord(data) && Array.isArray(data.choices) && isRecord(data.choices[0]) && isRecord(data.choices[0].message) + ? data.choices[0].message.content + : null + if (typeof content !== "string") { + throw new Error("Model response missing choices[0].message.content") + } + const parsed: unknown = JSON.parse(content) + return isRecord(parsed) ? parsed : {} +} + +const toStringMap = (raw: Record): Record => + Object.fromEntries(Object.entries(raw).map(([key, value]) => [key, typeof value === "string" ? value : String(value)])) + +const toNumberMap = (raw: Record): Record => + Object.fromEntries(Object.entries(raw).map(([key, value]) => [key, typeof value === "number" ? value : Number(value)])) + +export const runOpenAICompatible = async (config: OpenAICompatibleConfig, input: BenchInput): Promise => { + const fillable = input.fields.filter(isFillable) + const fieldLines = fillable + .map((field) => `- id=${field.fieldId} label="${field.name}" type=${field.type}${field.options !== null ? ` options=[${field.options.join(", ")}]` : ""}`) + .join("\n") + const recordLines = input.record.columns.map((column) => `- ${column} = "${input.record.values[column] ?? ""}"`).join("\n") + + const mapStart = performance.now() + const mapRaw = await chatJson( + config, + 'You fill PDF form fields from a CSV record. For each field, pick the CSV column whose value belongs in it, or "none". For a field with options, pick one of its options (or "none"). Reply ONLY with a JSON object mapping each field id to the chosen column/option/"none".', + `CSV record:\n${recordLines}\n\nForm fields:\n${fieldLines}`, + ) + const mapMs = performance.now() - mapStart + + const filledLines = input.filled.map((field) => `- label="${field.name}" value="${field.value}"`).join("\n") + const classifyStart = performance.now() + const plausRaw = await chatJson( + config, + "You judge whether each filled PDF field value is genuine and plausible (not fictional, a placeholder, or obviously wrong). Reply ONLY with a JSON object mapping each field label to a number from 0 (implausible) to 1 (genuine).", + `Filled fields:\n${filledLines}`, + ) + const classifyMs = performance.now() - classifyStart + + return { timing: { mapMs, classifyMs }, mapping: toStringMap(mapRaw), plausibility: toNumberMap(plausRaw) } +} diff --git a/examples/with-typesafe/benchmark/benchmark.test.ts b/examples/with-typesafe/benchmark/benchmark.test.ts new file mode 100644 index 0000000..7881354 --- /dev/null +++ b/examples/with-typesafe/benchmark/benchmark.test.ts @@ -0,0 +1,195 @@ +// Benchmark-only (delete benchmark/ before the PR). Times JEV vs an OpenAI-compatible model on the +// SAME W-9 payload: a mapping pass (map every field) + a plausibility pass (judge every filled +// value), each a single batched call, run repeatedly. It measures pure model latency — the editor +// round-trips are model-independent and excluded on purpose. Run: `npm run bench`. +// +// Env (in .env.local or the shell): +// NEXT_PUBLIC_TYPESAFE_API_KEY the Typesafe/JEV key (TYPESAFE_API_KEY also accepted) +// BENCHMARK_MODEL_BASE_URL OpenAI-compatible base that serves the model, e.g. https://api.deepseek.com +// BENCHMARK_MODEL_ID model id, e.g. deepseek_v4_flash +// BENCHMARK_MODEL_API_KEY that provider's key +// BENCHMARK_MODEL_EXTRA_BODY optional JSON merged into each request, e.g. {"reasoning_effort":"low"} +// BENCH_RUNS timed runs per model (default 5) +import { existsSync, readFileSync } from "node:fs" +import { join } from "node:path" +import { describe, expect, it } from "vitest" +import { W9_INPUT, type BenchInput } from "./fixtures" +import { runJev, runOpenAICompatible, type ModelOutput } from "./backends" + +// A field is flagged implausible below this score (matches the app's PLAUSIBILITY_MIN). +const PLAUSIBILITY_MIN = 0.3 + +// Populate process.env from .env.local for any key not already set (vitest, unlike Next, does not). +const loadEnvLocal = (): void => { + const path = join(process.cwd(), ".env.local") + if (!existsSync(path)) { + return + } + for (const line of readFileSync(path, "utf8").split("\n")) { + const trimmed = line.trim() + if (trimmed === "" || trimmed.startsWith("#")) { + continue + } + const separator = trimmed.indexOf("=") + if (separator === -1) { + continue + } + const key = trimmed.slice(0, separator).trim() + const value = trimmed.slice(separator + 1).trim().replace(/^["']|["']$/g, "") + if (process.env[key] === undefined) { + process.env[key] = value + } + } +} + +const nonEmpty = (value: string | undefined): string | null => (value !== undefined && value.trim() !== "" ? value : null) + +const isRecord = (value: unknown): value is Record => + typeof value === "object" && value !== null && !Array.isArray(value) + +const parseExtraBody = (raw: string | null): Record => { + if (raw === null) { + return {} + } + try { + const parsed: unknown = JSON.parse(raw) + return isRecord(parsed) ? parsed : {} + } catch { + return {} + } +} + +type Stats = { runs: number; avg: number; min: number; max: number; p50: number } +const stats = (values: number[]): Stats => { + const sorted = [...values].sort((a, b) => a - b) + const sum = sorted.reduce((total, value) => total + value, 0) + const count = sorted.length + return { + runs: count, + avg: count === 0 ? 0 : sum / count, + min: sorted[0] ?? 0, + max: sorted[count - 1] ?? 0, + p50: sorted[Math.floor(count / 2)] ?? 0, + } +} + +type ModelResult = { name: string; map: Stats; classify: Stats; total: Stats; sample: ModelOutput | undefined } + +const benchmarkModel = async ( + name: string, + run: (input: BenchInput) => Promise, + input: BenchInput, + runs: number, +): Promise => { + await run(input) // warm-up, discarded (excludes cold-start / connection setup) + const outputs: ModelOutput[] = [] + for (const _ of Array.from({ length: runs })) { + outputs.push(await run(input)) + } + const timings = outputs.map((output) => output.timing) + return { + name, + map: stats(timings.map((timing) => timing.mapMs)), + classify: stats(timings.map((timing) => timing.classifyMs)), + total: stats(timings.map((timing) => timing.mapMs + timing.classifyMs)), + sample: outputs[outputs.length - 1], + } +} + +// Are the two models producing the SAME decisions? Latency only means something once they do. +// "none", "unchecked" and "" all mean "leave the field blank", so they compare as equal. +const BLANK_DECISIONS = new Set(["none", "unchecked", ""]) +const canonical = (value: string | undefined): string => { + const normalized = (value ?? "").trim().toLowerCase() + return BLANK_DECISIONS.has(normalized) ? "" : normalized +} +const flagged = (score: number | undefined): boolean => score !== undefined && score < PLAUSIBILITY_MIN + +const compareOutputs = (a: ModelResult, b: ModelResult): void => { + const sampleA = a.sample + const sampleB = b.sample + if (sampleA === undefined || sampleB === undefined) { + return + } + console.log(`output agreement — ${a.name} vs ${b.name}:\n`) + + const fieldIds = Object.keys(sampleA.mapping) + const mapDisagreements = fieldIds.filter((id) => canonical(sampleA.mapping[id]) !== canonical(sampleB.mapping[id])) + console.log(` mapping: ${fieldIds.length - mapDisagreements.length}/${fieldIds.length} fields chose the same column/option (blank-equivalents collapsed)`) + for (const id of mapDisagreements) { + console.log(` - ${id}: ${a.name}="${sampleA.mapping[id] ?? "?"}" ${b.name}="${sampleB.mapping[id] ?? "?"}"`) + } + + const labels = Object.keys(sampleA.plausibility) + const plausDisagreements = labels.filter((label) => flagged(sampleA.plausibility[label]) !== flagged(sampleB.plausibility[label])) + console.log(`\n plausibility: ${labels.length - plausDisagreements.length}/${labels.length} filled fields agree on flagged-implausible (< ${PLAUSIBILITY_MIN})`) + for (const label of plausDisagreements) { + console.log(` - "${label}": ${a.name}=${(sampleA.plausibility[label] ?? 0).toFixed(2)} ${b.name}=${(sampleB.plausibility[label] ?? 0).toFixed(2)}`) + } + console.log("") +} + +const ms = (value: number): string => `${Math.round(value)}ms` + +const printComparison = (results: ModelResult[]): void => { + const rows = results.map((result) => ({ + name: result.name, + map: `${ms(result.map.avg)} (min ${ms(result.map.min)}, max ${ms(result.map.max)})`, + classify: `${ms(result.classify.avg)} (min ${ms(result.classify.min)}, max ${ms(result.classify.max)})`, + total: ms(result.total.avg), + })) + const width = (pick: (row: (typeof rows)[number]) => string, header: string): number => + Math.max(header.length, ...rows.map((row) => pick(row).length)) + const nameW = width((row) => row.name, "Model") + const mapW = width((row) => row.map, "map avg") + const classifyW = width((row) => row.classify, "classify avg") + const line = (name: string, map: string, classify: string, total: string): string => + `${name.padEnd(nameW)} ${map.padEnd(mapW)} ${classify.padEnd(classifyW)} ${total}` + + console.log(`\nfill + classify latency, ${results[0]?.total.runs ?? 0} timed runs each (warm-up discarded):\n`) + console.log(line("Model", "map avg", "classify avg", "total avg")) + for (const row of rows) { + console.log(line(row.name, row.map, row.classify, row.total)) + } + if (results.length === 2) { + const [first, second] = results + if (first !== undefined && second !== undefined) { + const faster = first.total.avg <= second.total.avg ? first : second + const slower = faster === first ? second : first + const ratio = slower.total.avg / Math.max(faster.total.avg, 1) + console.log(`\n→ ${faster.name} is ${ratio.toFixed(2)}× faster end-to-end (${ms(faster.total.avg)} vs ${ms(slower.total.avg)}).`) + } + } + console.log("") +} + +describe("JEV vs OpenAI-compatible fill + classify latency", () => { + it("times both models on the W-9 payload", async () => { + loadEnvLocal() + const runs = Number(nonEmpty(process.env.BENCH_RUNS) ?? "5") + const results: ModelResult[] = [] + + const typesafeKey = nonEmpty(process.env.NEXT_PUBLIC_TYPESAFE_API_KEY) ?? nonEmpty(process.env.TYPESAFE_API_KEY) + if (typesafeKey !== null) { + results.push(await benchmarkModel("JEV (System One)", (input) => runJev(typesafeKey, input), W9_INPUT, runs)) + } else { + console.log("skipping JEV: set NEXT_PUBLIC_TYPESAFE_API_KEY (or TYPESAFE_API_KEY)") + } + + const baseUrl = nonEmpty(process.env.BENCHMARK_MODEL_BASE_URL) + const modelId = nonEmpty(process.env.BENCHMARK_MODEL_ID) + const modelKey = nonEmpty(process.env.BENCHMARK_MODEL_API_KEY) + if (baseUrl !== null && modelId !== null && modelKey !== null) { + const config = { baseUrl, model: modelId, apiKey: modelKey, extraBody: parseExtraBody(nonEmpty(process.env.BENCHMARK_MODEL_EXTRA_BODY)) } + results.push(await benchmarkModel(modelId, (input) => runOpenAICompatible(config, input), W9_INPUT, runs)) + } else { + console.log("skipping second model: set BENCHMARK_MODEL_BASE_URL, BENCHMARK_MODEL_ID, BENCHMARK_MODEL_API_KEY") + } + + printComparison(results) + if (results.length === 2 && results[0] !== undefined && results[1] !== undefined) { + compareOutputs(results[0], results[1]) + } + expect(results.length).toBeGreaterThan(0) + }, 600_000) +}) diff --git a/examples/with-typesafe/benchmark/fixtures.ts b/examples/with-typesafe/benchmark/fixtures.ts new file mode 100644 index 0000000..04b6627 --- /dev/null +++ b/examples/with-typesafe/benchmark/fixtures.ts @@ -0,0 +1,70 @@ +// Benchmark-only fixtures (delete with the rest of benchmark/ before the PR). A faithful stand-in +// for the W-9's get_fields output + demo record, so both models are timed on the same payload +// without needing the live editor. The exact field ids don't affect latency; the field COUNT, +// labels, types and CSV size do, and those mirror the shipped W-9 demo. + +export type BenchFieldType = "TEXT" | "SIGNATURE" | "PICTURE" | "CHECKBOX" | "COMB_TEXT" | "DROPDOWN" | "RADIO" +export type BenchField = { fieldId: string; name: string; type: BenchFieldType; options: string[] | null } +export type BenchRecord = { columns: string[]; values: Record } +export type BenchFilledField = { name: string; value: string } +export type BenchInput = { fields: BenchField[]; record: BenchRecord; filled: BenchFilledField[] } + +const CHECKBOX_OPTIONS = ["checked", "xchecked", "unchecked"] + +const text = (fieldId: string, name: string): BenchField => ({ fieldId, name, type: "TEXT", options: null }) +const comb = (fieldId: string, name: string): BenchField => ({ fieldId, name, type: "COMB_TEXT", options: null }) +const checkbox = (fieldId: string, name: string): BenchField => ({ fieldId, name, type: "CHECKBOX", options: CHECKBOX_OPTIONS }) + +const W9_FIELDS: BenchField[] = [ + text("f0", "Name of entity / individual"), + text("f1", "Business name"), + checkbox("f2", "Individual / sole proprietor"), + checkbox("f3", "C corporation"), + checkbox("f4", "S corporation"), + checkbox("f5", "Partnership"), + checkbox("f6", "Trust / estate"), + checkbox("f7", "LLC"), + text("f8", "LLC Tax classification"), + text("f9", "Other (entity name as per instructions)"), + checkbox("f10", "Other (see instructions)"), + text("f11", "Exempt payee code (if any)"), + text("f12", "Exemption from FATCA reporting code (if any)"), + checkbox("f13", "Foreign partners, owners, or beneficiaries (see instructions)"), + text("f14", "Address (number, street, and apt. or suite no.)"), + text("f15", "City, state, and ZIP code"), + text("f16", "Requester's name and address (optional)"), + text("f17", "List account number(s) here (optional)"), + comb("f18", "SSN (first 3 digits)"), + comb("f19", "SSN (middle 2 digits)"), + comb("f20", "SSN (last 4 digits)"), + comb("f21", "EIN (first 2 digits)"), + comb("f22", "EIN (last 7 digits)"), + { fieldId: "f23", name: "signature", type: "SIGNATURE", options: null }, +] + +const W9_RECORD: BenchRecord = { + columns: ["name", "business_name", "federal_tax_classification", "address", "city_state_zip", "taxpayer_id"], + values: { + name: "Northwind Labs, Inc.", + business_name: "Northwind Labs", + federal_tax_classification: "C Corporation", + address: "500 Howard Street", + city_state_zip: "San Francisco, CA 94105", + taxpayer_id: "84-1234567", + }, +} + +// The values that would be filled after the mapping pass, judged by the plausibility pass. Fixed +// so both models judge the identical set (decoupled from each model's own mapping result). +const W9_FILLED: BenchFilledField[] = [ + { name: "Name of entity / individual", value: "Northwind Labs, Inc." }, + { name: "Business name", value: "Northwind Labs" }, + { name: "LLC Tax classification", value: "C Corporation" }, + { name: "Address (number, street, and apt. or suite no.)", value: "500 Howard Street" }, + { name: "City, state, and ZIP code", value: "San Francisco, CA 94105" }, + { name: "SSN (first 3 digits)", value: "84-1234567" }, + { name: "EIN (first 2 digits)", value: "84-1234567" }, + { name: "EIN (last 7 digits)", value: "84-1234567" }, +] + +export const W9_INPUT: BenchInput = { fields: W9_FIELDS, record: W9_RECORD, filled: W9_FILLED } diff --git a/examples/with-typesafe/components.json b/examples/with-typesafe/components.json new file mode 100644 index 0000000..32fd4a6 --- /dev/null +++ b/examples/with-typesafe/components.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://ui.shadcn.com/schema.json", + "style": "default", + "rsc": true, + "tsx": true, + "tailwind": { + "config": "", + "css": "app/globals.css", + "baseColor": "neutral", + "cssVariables": true, + "prefix": "" + }, + "aliases": { + "components": "@/components", + "utils": "@/lib/utils", + "ui": "@/components/ui", + "lib": "@/lib", + "hooks": "@/hooks" + }, + "iconLibrary": "lucide" +} \ No newline at end of file diff --git a/examples/with-typesafe/components/form-gallery.tsx b/examples/with-typesafe/components/form-gallery.tsx new file mode 100644 index 0000000..70e6952 --- /dev/null +++ b/examples/with-typesafe/components/form-gallery.tsx @@ -0,0 +1,36 @@ +"use client" + +import type { ReactElement } from "react" +import type { FormDefinition } from "@/lib/forms" + +type FormGalleryProps = { + forms: FormDefinition[] + activeFormId: string + onSelectForm: (form: FormDefinition) => void + disabled: boolean +} + +export const FormGallery = ({ forms, activeFormId, onSelectForm, disabled }: FormGalleryProps): ReactElement => ( +
+ Form +
+ {forms.map((form) => { + const isActive = form.id === activeFormId + return ( + + ) + })} +
+
+) diff --git a/examples/with-typesafe/components/source-pane.tsx b/examples/with-typesafe/components/source-pane.tsx new file mode 100644 index 0000000..0b92fcd --- /dev/null +++ b/examples/with-typesafe/components/source-pane.tsx @@ -0,0 +1,287 @@ +"use client" + +import type { ReactElement } from "react" +import { Button } from "@/components/ui/button" +import { FormGallery } from "@/components/form-gallery" +import { Timing, type TimingPhase } from "@/components/timing" +import type { ParsedCsvRecord } from "@/lib/csv" +import type { FormDefinition } from "@/lib/forms" +import { + buildFieldRows, + primaryAction, + type Issue, + type FillSummary, + type FieldMapping, + type EditorField, + type FieldRowState, + type PrimaryAction, +} from "@/lib/jev" + +const primaryActionLabel = (action: PrimaryAction): string => { + switch (action) { + case "fill": + return "Fill & validate" + case "validate": + return "Validate" + case "revalidate": + return "Re-validate" + default: + action satisfies never + return "" + } +} + +const rowIndicator = (state: FieldRowState): { className: string; symbol: string } => { + switch (state) { + case "filled": + return { className: "text-valid", symbol: "●" } + case "confirm": + return { className: "text-warn", symbol: "▲" } + case "confirmed": + return { className: "text-valid", symbol: "●" } + case "error": + return { className: "text-error", symbol: "✕" } + case "empty": + return { className: "text-muted-foreground/40", symbol: "·" } + default: + state satisfies never + return { className: "text-muted-foreground", symbol: "·" } + } +} + +type SourcePaneProps = { + forms: FormDefinition[] + activeFormId: string + onSelectForm: (form: FormDefinition) => void + record: ParsedCsvRecord | null + mappings: FieldMapping[] + fields: EditorField[] + issues: Issue[] + fillSummary: FillSummary | null + errorMessage: string | null + onFillAndValidate: () => void + canFillAndValidate: boolean + isBusy: boolean + validationStale: boolean + timingPhase: TimingPhase + fillMs: number | null + classifyMs: number | null + onSelectField: (fieldId: string, page: number) => void + onApplyCandidate: (issue: Issue, column: string) => void + onConfirmMapping: (fieldId: string) => void + onFinalize: () => void + confirmedFieldIds: ReadonlySet +} + +export const SourcePane = ({ + forms, + activeFormId, + onSelectForm, + record, + mappings, + fields, + issues, + fillSummary, + errorMessage, + onFillAndValidate, + canFillAndValidate, + isBusy, + validationStale, + timingPhase, + fillMs, + classifyMs, + onSelectField, + onApplyCandidate, + onConfirmMapping, + onFinalize, + confirmedFieldIds, +}: SourcePaneProps): ReactElement => { + const isFilled = fillSummary !== null + const fieldRows = isFilled ? buildFieldRows(fields, mappings, issues, confirmedFieldIds) : [] + // Any actionable field (a value, a confirmed decision, or an open issue to confirm/fix) lives in + // "Fields". Only a still-empty, untouched, unflagged field is "Additional" — so the Additional + // section never holds anything the human must act on. A manually filled field bumps to Fields on + // the next re-validate (which re-reads its value). + const mainRows = fieldRows.filter((row) => row.state !== "empty") + const additionalRows = fieldRows.filter((row) => row.state === "empty") + const filledCount = mainRows.filter((row) => row.state === "filled").length + const errorCount = mainRows.filter((row) => row.state === "error").length + const confirmCount = mainRows.filter((row) => row.state === "confirm").length + const isResolved = isFilled && errorCount === 0 && confirmCount === 0 && !validationStale + + const renderFieldRow = (row: (typeof fieldRows)[number]): ReactElement => { + const indicator = rowIndicator(row.state) + const issue = row.issue + // Every amber issue is confirmable, whether it came from a medium-confidence mapping or a + // plausibility flag: confirming accepts the current editor value as the human's decision. + // Red issues are hard validation errors and must be fixed, not confirmed away. + const isConfirmable = issue !== null && issue.severity === "amber" + // Candidate columns are an alternative to a fill, so they only help while the field is still + // empty. Once filled, the confirm checkbox is the only action left. + const showCandidates = issue !== null && issue.candidates.length > 0 && row.value.trim() === "" + // An empty flagged field (a rejected write, or no confident column) is one the human fills by + // hand: offer a jump-to-field button. A filled amber only needs confirming, not manual entry. + const needsManualEntry = issue !== null && row.value.trim() === "" + return ( +
+ + {isConfirmable ? ( + + ) : null} + {row.sourceColumn !== null && row.value.trim() !== "" ? ( +

+ from{" "} + {row.fromNote !== null ? ( + + {row.sourceColumn} ({row.fromNote}) + + ) : ( + row.sourceColumn + )} +

+ ) : null} + {issue !== null && (issue.message !== "" || showCandidates || needsManualEntry) ? ( +
+ {issue.message !== "" ? ( +

{issue.message}

+ ) : null} + {showCandidates || needsManualEntry ? ( +
+ {issue.candidates.map((candidate) => ( + + ))} + {needsManualEntry ? ( + + ) : null} +
+ ) : null} +
+ ) : null} +
+ ) + } + + return ( + + ) +} diff --git a/examples/with-typesafe/components/timing.tsx b/examples/with-typesafe/components/timing.tsx new file mode 100644 index 0000000..ec9447e --- /dev/null +++ b/examples/with-typesafe/components/timing.tsx @@ -0,0 +1,68 @@ +"use client" + +import { useEffect, useState, type ReactElement } from "react" + +export type TimingPhase = "fill" | "classify" | null + +const Chronometer = ({ + label, + explanation, + running, + durationMs, +}: { + label: string + explanation: string + running: boolean + durationMs: number | null +}): ReactElement => { + const [liveMs, setLiveMs] = useState(0) + + // A live count-up needs a ticking clock; the rare justified useEffect. A 100ms interval + // is plenty for an integer-ms display and avoids ~60 renders/sec over a multi-second run. + useEffect(() => { + if (!running) { + return + } + const start = performance.now() + const intervalId = setInterval(() => setLiveMs(performance.now() - start), 100) + return () => clearInterval(intervalId) + }, [running]) + + const shown = running ? liveMs : durationMs + + return ( +
+ {label} + + {shown === null ? "—" : Math.round(shown)} + ms + + {explanation} +
+ ) +} + +export const Timing = ({ + phase, + fillMs, + classifyMs, +}: { + phase: TimingPhase + fillMs: number | null + classifyMs: number | null +}): ReactElement => ( +
+ + +
+) diff --git a/examples/with-typesafe/components/ui/button.tsx b/examples/with-typesafe/components/ui/button.tsx new file mode 100644 index 0000000..0983e3d --- /dev/null +++ b/examples/with-typesafe/components/ui/button.tsx @@ -0,0 +1,41 @@ +import { Slot } from "@radix-ui/react-slot" +import { cva, type VariantProps } from "class-variance-authority" +import type { ButtonHTMLAttributes, ReactElement, Ref } from "react" + +import { cn } from "@/lib/utils" + +const buttonVariants = cva( + "inline-flex items-center justify-center gap-2 whitespace-nowrap rounded-md text-sm font-medium ring-offset-background transition-colors focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:pointer-events-none disabled:opacity-50 [&_svg]:pointer-events-none [&_svg]:size-4 [&_svg]:shrink-0", + { + variants: { + variant: { + default: "bg-primary text-primary-foreground hover:bg-primary/90", + destructive: "bg-destructive text-destructive-foreground hover:bg-destructive/90", + outline: "border border-input bg-background hover:bg-accent hover:text-accent-foreground", + secondary: "bg-secondary text-secondary-foreground hover:bg-secondary/80", + ghost: "hover:bg-accent hover:text-accent-foreground", + link: "text-primary underline-offset-4 hover:underline", + }, + size: { + default: "h-10 px-4 py-2", + sm: "h-9 rounded-md px-3", + lg: "h-11 rounded-md px-8", + icon: "h-10 w-10", + }, + }, + defaultVariants: { + variant: "default", + size: "default", + }, + }, +) + +// A native