From 0131e65e10c6e612f339e6c6978c8fc81e8d829f Mon Sep 17 00:00:00 2001 From: suharshit singh Date: Wed, 16 Sep 2026 01:38:25 +0530 Subject: [PATCH] feat(spec): generate a Markdown technical spec from the canvas The Specs tab now turns the current canvas into a Markdown technical spec that downloads automatically: an overview, key decisions highlighted as callouts, one Mermaid diagram section per group of connected components, a component reference, connections, and risks. - spec-agent Trigger task: reads the room, builds the spec graph in code, makes one structured model call for the prose (sanitised against the graph, internal refs stripped), renders the Markdown in code; aborts on an empty or oversized canvas - Stored on read like AI turns: GET /api/projects/[id]/spec uploads a finished run to private Vercel Blob and records it on Project (guarded on the run id); POST starts a run (409 while one is pending); GET .../spec/markdown downloads the file - Decisions recorded in the user's AI design sessions are included and tagged as recorded - Migration: TaskRun.kind (DESIGN or SPEC) and Project spec fields - useSpecGenerator and SpecPanel replace the placeholder Specs tab - Shared lib/ai/model.ts, lib/ai/run-failure.ts and lib/canvas-room.ts used by both agents Co-Authored-By: Claude Opus 5 --- .../[projectId]/spec/markdown/route.ts | 36 +++ app/api/projects/[projectId]/spec/route.ts | 58 ++++ components/editor/ai-sidebar.tsx | 40 +-- components/editor/spec-panel.tsx | 100 +++++++ context/architecture-context.md | 20 +- context/progress-tracker.md | 79 +++++- context/project-overview.md | 13 +- context/ui-context.md | 3 +- hooks/use-spec-generator.ts | 236 ++++++++++++++++ lib/ai/design-agent-engine.ts | 55 +--- lib/ai/model.ts | 54 ++++ lib/ai/run-failure.ts | 35 +++ lib/ai/session-turns.ts | 28 +- lib/canvas-room.ts | 15 + lib/spec/mermaid.ts | 69 +++++ lib/spec/render-markdown.ts | 229 +++++++++++++++ lib/spec/spec-engine.ts | 38 +++ lib/spec/spec-file-name.ts | 11 + lib/spec/spec-graph.ts | 222 +++++++++++++++ lib/spec/spec-prompt.ts | 98 +++++++ lib/spec/spec-schema.ts | 247 ++++++++++++++++ lib/spec/spec-store.ts | 265 ++++++++++++++++++ .../migration.sql | 14 + prisma/models/project.prisma | 5 + prisma/models/task-run.prisma | 15 +- src/trigger/design-agent.ts | 10 +- src/trigger/spec-agent.ts | 67 +++++ 27 files changed, 1939 insertions(+), 123 deletions(-) create mode 100644 app/api/projects/[projectId]/spec/markdown/route.ts create mode 100644 app/api/projects/[projectId]/spec/route.ts create mode 100644 components/editor/spec-panel.tsx create mode 100644 hooks/use-spec-generator.ts create mode 100644 lib/ai/model.ts create mode 100644 lib/ai/run-failure.ts create mode 100644 lib/canvas-room.ts create mode 100644 lib/spec/mermaid.ts create mode 100644 lib/spec/render-markdown.ts create mode 100644 lib/spec/spec-engine.ts create mode 100644 lib/spec/spec-file-name.ts create mode 100644 lib/spec/spec-graph.ts create mode 100644 lib/spec/spec-prompt.ts create mode 100644 lib/spec/spec-schema.ts create mode 100644 lib/spec/spec-store.ts create mode 100644 prisma/migrations/20260915195650_add_project_spec/migration.sql create mode 100644 src/trigger/spec-agent.ts diff --git a/app/api/projects/[projectId]/spec/markdown/route.ts b/app/api/projects/[projectId]/spec/markdown/route.ts new file mode 100644 index 0000000..e8c5a3f --- /dev/null +++ b/app/api/projects/[projectId]/spec/markdown/route.ts @@ -0,0 +1,36 @@ +import { auth } from "@clerk/nextjs/server"; + +import { getAccessibleProject, getCurrentIdentity } from "@/lib/project-access"; +import { toSpecFileName } from "@/lib/spec/spec-file-name"; +import { readSpecMarkdown } from "@/lib/spec/spec-store"; + +type SpecMarkdownRouteContext = { params: Promise<{ projectId: string }> }; + +/** Downloads the project's latest spec as a Markdown file. */ +export async function GET(_request: Request, context: SpecMarkdownRouteContext) { + const { userId } = await auth(); + if (!userId) { + return Response.json({ error: "Unauthorized" }, { status: 401 }); + } + + const { projectId } = await context.params; + const { primaryEmail } = await getCurrentIdentity(); + const project = await getAccessibleProject(userId, primaryEmail, projectId); + if (!project) { + return Response.json({ error: "Forbidden" }, { status: 403 }); + } + + const markdown = await readSpecMarkdown(project.id); + if (markdown === null) { + return Response.json({ error: "No spec has been generated yet" }, { status: 404 }); + } + + return new Response(markdown, { + status: 200, + headers: { + "Content-Type": "text/markdown; charset=utf-8", + "Content-Disposition": `attachment; filename="${toSpecFileName(project.name)}"`, + "Cache-Control": "no-store", + }, + }); +} diff --git a/app/api/projects/[projectId]/spec/route.ts b/app/api/projects/[projectId]/spec/route.ts new file mode 100644 index 0000000..df6ed8f --- /dev/null +++ b/app/api/projects/[projectId]/spec/route.ts @@ -0,0 +1,58 @@ +import { auth } from "@clerk/nextjs/server"; + +import { getAccessibleProject, getCurrentIdentity } from "@/lib/project-access"; +import { getSpecStatus, startSpecRun } from "@/lib/spec/spec-store"; + +type SpecRouteContext = { params: Promise<{ projectId: string }> }; + +async function resolveProject(projectId: string) { + const { userId } = await auth(); + if (!userId) { + return { error: Response.json({ error: "Unauthorized" }, { status: 401 }) } as const; + } + + const { primaryEmail } = await getCurrentIdentity(); + const project = await getAccessibleProject(userId, primaryEmail, projectId); + if (!project) { + return { error: Response.json({ error: "Forbidden" }, { status: 403 }) } as const; + } + + return { project, userId } as const; +} + +/** + * The project's spec status: the stored spec (if any), a pending run, or why + * the latest run failed. A finished run is stored before responding. + */ +export async function GET(_request: Request, context: SpecRouteContext) { + const { projectId } = await context.params; + const result = await resolveProject(projectId); + if ("error" in result) { + return result.error; + } + + return Response.json(await getSpecStatus(result.project.id)); +} + +/** Starts generating a spec from the current canvas. Owners and collaborators can generate. */ +export async function POST(_request: Request, context: SpecRouteContext) { + const { projectId } = await context.params; + const result = await resolveProject(projectId); + if ("error" in result) { + return result.error; + } + + const started = await startSpecRun({ + projectId: result.project.id, + userId: result.userId, + projectName: result.project.name, + }); + + if (started.ok) { + return Response.json({ runId: started.runId }, { status: 202 }); + } + if (started.status === 409) { + return Response.json({ error: started.error, runId: started.runId }, { status: 409 }); + } + return Response.json({ error: started.error }, { status: started.status }); +} diff --git a/components/editor/ai-sidebar.tsx b/components/editor/ai-sidebar.tsx index b77f231..254bbbe 100644 --- a/components/editor/ai-sidebar.tsx +++ b/components/editor/ai-sidebar.tsx @@ -2,11 +2,13 @@ import { FormEvent, KeyboardEvent, useEffect, useRef, useState } from "react"; import { Tabs as TabsPrimitive } from "@base-ui/react/tabs"; -import { ArrowRight, ChevronDown, Download, FileText, History, Loader2, Plus, Sparkle, X } from "lucide-react"; +import { ArrowRight, ChevronDown, History, Loader2, Plus, Sparkle, X } from "lucide-react"; import { PlanCard, QuestionsCard, ResultCard } from "@/components/editor/ai-chat-cards"; import { AiSessionHistory } from "@/components/editor/ai-session-history"; +import { SpecPanel } from "@/components/editor/spec-panel"; import { useAiSession, type AiChatMessage } from "@/hooks/use-ai-session"; +import { useSpecGenerator } from "@/hooks/use-spec-generator"; import { readAnswersPayload, readPlanPayload, @@ -77,6 +79,8 @@ export function AiSidebar({ open, onClose, projectId }: AiSidebarProps) { removeSession, retryLoadSession, } = useAiSession(projectId); + // Lives here rather than in the Specs panel so a run keeps being followed (and downloads) while another tab is open. + const specGenerator = useSpecGenerator(projectId); const transcriptEndRef = useRef(null); @@ -394,39 +398,7 @@ export function AiSidebar({ open, onClose, projectId }: AiSidebarProps) { -
- - -
-
- -
-

Realtime Chat Platform Spec

-

- Includes service boundaries, event flow, storage strategy, and deployment notes. -

-
-
- -
-
+
diff --git a/components/editor/spec-panel.tsx b/components/editor/spec-panel.tsx new file mode 100644 index 0000000..2f67b53 --- /dev/null +++ b/components/editor/spec-panel.tsx @@ -0,0 +1,100 @@ +"use client"; + +import { Download, FileText, Loader2 } from "lucide-react"; + +import type { UseSpecGeneratorResult } from "@/hooks/use-spec-generator"; +import { cn } from "@/lib/utils"; + +const focusClass = "outline-none focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-ink/60"; + +const kickerClass = "font-mono text-chrome tracking-chrome text-ink-soft uppercase"; + +const dateFormatter = new Intl.DateTimeFormat("en", { dateStyle: "medium", timeStyle: "short" }); + +function plural(count: number, noun: string): string { + return `${count} ${noun}${count === 1 ? "" : "s"}`; +} + +/** The AI sidebar's Specs tab: generate a Markdown spec from the canvas and download it. */ +export function SpecPanel({ spec, isLoading, isRunning, statusText, error, downloadHref, generate }: UseSpecGeneratorResult) { + return ( +
+

+ Turn the current canvas into a Markdown technical spec: diagrams, key decisions, components, connections, and + risks. +

+ + + + {isRunning ? ( +

+ {statusText} +

+ ) : ( +

Downloads automatically when ready

+ )} + + {error ? ( +
+ {error} +
+ ) : null} + + {isLoading ? ( +

+

+ ) : spec ? ( +
+
+
+ + +
+ ) : ( +

+ No spec yet. Generate one from the current canvas. +

+ )} +
+ ); +} diff --git a/context/architecture-context.md b/context/architecture-context.md index f0a633f..ab94ec9 100644 --- a/context/architecture-context.md +++ b/context/architecture-context.md @@ -48,6 +48,23 @@ failure messages with raw provider errors kept in server logs. - **Model:** `GOOGLE_GENERATIVE_AI_MODEL`, with a thinking level (Gemini 3+) or thinking budget (Gemini 2.x) per step. +## Spec Agent + +- **One spec = one Trigger.dev run** (`spec-agent`). `POST /api/projects/[projectId]/spec` starts it (409 with the + pending run id if one is already running) and records a `TaskRun` with `kind: SPEC`. +- **The run** reads the room read-only, builds the spec graph in code (`lib/spec/spec-graph.ts`: groups of connected + components, refs, notes), makes one structured model call for the prose (`lib/spec/spec-engine.ts`, validated and + sanitised against the graph), and renders the Markdown in code (`lib/spec/render-markdown.ts`, Mermaid from + `lib/spec/mermaid.ts`). It aborts with a user-facing message on an empty canvas or more than 200 components. +- **Stored on read.** `GET /api/projects/[projectId]/spec` retrieves the latest spec run; when it has finished and is + not yet stored, the Markdown is uploaded to private Vercel Blob storage and `Project.specMdPath`, `specRunId`, + `specGeneratedAt`, and `specStats` are updated in a write guarded on the run id. The previous blob is deleted. + `GET /api/projects/[projectId]/spec/markdown` serves the file as a download. +- **Recorded decisions** from the requesting user's unexpired AI sessions (their `RESULT` payloads) are passed to the + run so the spec can highlight them. +- Model configuration, thinking settings, schema retry, and friendly failure messages are shared with the design agent + (`lib/ai/model.ts`, `lib/ai/run-failure.ts`). + ## Auth and Access Model - Every user signs in via Clerk to establish identity. @@ -61,7 +78,8 @@ 3. API routes must validate authentication and resource ownership before any mutation. 4. Foundation components (`components/ui/*`) must remain generic and default. 5. Generated canvas content reaches clients only through Liveblocks, never through an HTTP response. -6. The design task never writes to the database or blob storage; the session routes store AI turns. +6. Background AI tasks (design and spec agents) never write to the database or blob storage; the session and spec + routes store their results. 7. A project's id is its Liveblocks room id. 8. Run tokens are minted only by `POST /api/ai/design/token`, after a `TaskRun` ownership check. 9. Model output is validated (zod) before it is stored or drawn. diff --git a/context/progress-tracker.md b/context/progress-tracker.md index 32e4e1d..770d863 100644 --- a/context/progress-tracker.md +++ b/context/progress-tracker.md @@ -11,9 +11,11 @@ Update this file whenever the current phase, active feature, or implementation s - Persistent, multi-turn AI design sessions (plan: storage → wire sessions → turn engine → UI cards → generation quality → docs). Steps 1 (storage, PR #21), 2 (sessions in the sidebar, PR #22) and 3 (clarify → plan → generate turn engine, PR #23) and 4 (question/plan/result cards, PR #25) are done; step 5 (generation quality: canvas - context, roles and kickers, async edges, validate-and-repair, PR #26) is done; step 6 (docs) is on - `feat/ai-sessions-docs`. Next: the Specs tab (Generate Spec + automatic Markdown download), which can reuse the - plan decisions stored in RESULT payloads. + context, roles and kickers, async edges, validate-and-repair, PR #26) and step 6 (docs, PR #27) are done. +- Generate Spec (unit G1) is built on `feat/generate-spec`, stacked on #27, and awaiting review. +- Repository note (2026-09-16): GitHub `main` contains #21–#23 only. #25 and #26 were merged into their stacked + base branches after those bases had already merged, so steps 4 and 5 are not on `main` yet; #27 and the G1 PR + carry them. - After that: the Specs tab (Generate Spec + automatic Markdown download), which remains inert. ## Completed @@ -1466,3 +1468,74 @@ Update this file whenever the current phase, active feature, or implementation s - Section order after the splice: Getting started, Platform, AI, Authentication, Design system, Reference, Contributing, Roadmap. - No code changed, so lint, typecheck, and build were not re-run. +- Generate Spec, unit G1: Markdown technical spec from the canvas, with automatic download (2026-09-16, branch + `feat/generate-spec`, based on `feat/ai-sessions-docs` because GitHub `main` is missing #25 and #26): + - Data (migration `20260915195650_add_project_spec`, additive, applied): `TaskRunKind` enum (`DESIGN`, `SPEC`) and + `TaskRun.kind` (default `DESIGN`, so existing rows are design runs) with a `(projectId, kind, createdAt)` index; + `Project.specMdPath`, `specRunId`, `specGeneratedAt`, `specStats` (JSON). + - Shared refactors: `lib/ai/model.ts` (model id, thinking level/budget, `withSchemaRetry`, `modelIdOf`) moved out of + `design-agent-engine.ts`; `lib/ai/run-failure.ts` (`describeRunFailure` with generic, timeout and passthrough + messages) used by `session-turns.ts` and the spec store; `lib/canvas-room.ts` (`readRoomSnapshot`) used by both + tasks. No behaviour change for design turns. + - `lib/spec/spec-graph.ts`: `buildSpecGraph` — drawn, labelled components (cap 200, total kept), connections + between them (unknown endpoints and self loops dropped), groups of connected components via union-find in + reading order (`g1`…) plus a `standalone` group, component refs (`c1`…) in group then position order, kinds from + kicker or shape, and text notes attached to the nearest group within 600px (otherwise general notes). + - `lib/spec/mermaid.ts`: `renderGroupMermaid` — `flowchart TD`, one node per component with shape mapping (pill + `([ ])`, rectangle `[ ]`, cylinder `[( )]`, circle `(( ))`, diamond `{ }`, hexagon `{{ }}`), quoted labels with + `#`, `"`, `<`, `>`, `|` escaped, arrows for forward / backward (reversed) / bidirectional / none, dashed for + async and dotted edges, quoted edge labels. + - `lib/spec/spec-schema.ts`: model output schema (overview, goals, group name/purpose/flow, component + responsibilities, decisions with alternatives, trade-offs, component refs and source + `recorded | stated | inferred`, risks with mitigations, open questions); `sanitizeSpecContent` trims to list + limits, drops unknown groups/components/refs, and removes internal refs such as "(c1)" from prose (a live run + showed flow steps like "Web Client (c1) sends…"); run result and status types; stage names; abort messages. + - `lib/spec/spec-prompt.ts` + `spec-engine.ts`: one `generateText` + `Output.object` call with medium thinking, a + 180s timeout, and one schema retry. The prompt lists groups, refs, connections (direction, label, sync/async), + notes, and recorded decisions, and asks for 3–6 decisions from the diagram in addition to recorded ones (the + first live run returned only the recorded decision). + - `lib/spec/render-markdown.ts`: title and generation line, contents, 1 Overview (goals, general notes), 2 Key + decisions as `> [!IMPORTANT]` callouts (why, alternatives, trade-offs, components, source), 3 System diagrams + (per group: purpose, Mermaid, component table, flow, notes; "Standalone components"), 4 Component reference + (kind, diagram, responsibility, receives from / sends to / connected to), 5 Connections table, 6 Risks table and + open-question checklist, footer with model and time. Table cells escape pipes; fallbacks when prose is missing. + - `src/trigger/spec-agent.ts`: reads the room, aborts with a user-facing message on an empty canvas or more than 200 + components, generates the prose, renders, returns `{ markdown, stats }`. Single attempt, `maxDuration` 300s, + stages `reading` → `writing` → `rendering` → `done`. + - `lib/spec/spec-store.ts` (Trigger and Blob injectable for tests): `getSpecStatus` stores the latest finished SPEC + run on read (upload to private Blob, update guarded on `specRunId`, delete the duplicate or previous blob) and + reports pending runs and friendly failures; `loadRecordedDecisions` (the user's unexpired completed RESULT + messages, newest first, one per title, max 12); `startSpecRun` (409 with the pending run, trigger, `TaskRun` kind + SPEC); `readSpecMarkdown`. + - Routes: `GET`/`POST /api/projects/[projectId]/spec` (status; start → 202 / 409 / 502) and + `GET /api/projects/[projectId]/spec/markdown` (attachment `{project-slug}-spec.md`, via + `lib/spec/spec-file-name.ts` because route files may only export handlers). Owner or collaborator access. + - `hooks/use-spec-generator.ts` (mounted in `AiSidebar` so a run is followed while another tab is open): loads the + status, starts runs (409 follows the existing run), Realtime stages via the existing token route plus 4s + polling, and downloads automatically when the run the user started is stored. + - `components/editor/spec-panel.tsx`: description, Generate / Regenerate button with spinner and stage text, error + alert, latest-spec card (generated time, counts, **Download .md** link), empty and loading states. Replaces the + placeholder Specs tab in `ai-sidebar.tsx`. + - Context: `project-overview.md` Spec Generation describes the built behaviour; `architecture-context.md` gains a + Spec Agent section and invariant 6 covers both tasks; `ui-context.md` lists the Specs tab. `docs.md` (local, + gitignored) Spec generation page, reference tables, roadmap, decisions, and history updated. + - Validation checks: + - `pnpm typecheck` and `pnpm lint` passed (after `next typegen` for the new routes) + - Offline spec checks (60, no model or DB): grouping, refs, kinds, dropped edges, note placement, 200 cap; Mermaid + shapes, arrows, escaping; sanitising and limits; Markdown sections, callouts, tables, directions, notes, + fallbacks, pipe escaping; prompt text; file names; all 13 starter templates rendered with a stub spec (one + diagram per group, every component referenced, Mermaid line counts match) + - Ref stripping checks (10): refs removed from every prose field, paragraph breaks kept, other parentheses kept + - Spec store checks against the database with fake Trigger/Blob (22): empty, design runs ignored, pending, quota / + abort / unknown / missing failures, invalid output rejected before upload, stored once with finish time and + stats, not re-retrieved, failed newer run keeps the old spec, concurrent reads store once and delete the + duplicate and previous blobs, recorded decisions deduplicated and scoped to the user's unexpired completed + results + - Live runs through the local Trigger worker on `gemini-3.5-flash-lite` with the Microservices template and a + recorded decision (throwaway project, room, session, and blob removed): 202 → 409 for a second start → stored in + about 11s with stats 8 / 8 / 1; Markdown in Blob with all sections, one Mermaid diagram, every component, the + recorded decision tagged "Recorded during AI design". After the prompt change the second run also inferred an + "API Gateway Entry Point" decision (2 decisions, 0 warnings). + - Signed-out requests to the three spec routes are stopped by Clerk (307). + - Not yet verified: the Specs tab in a signed-in browser (button, stages, auto-download, Download link), Mermaid + rendering on GitHub, and the ref-stripping fix in a live run. diff --git a/context/project-overview.md b/context/project-overview.md index c3f5549..b1ce550 100644 --- a/context/project-overview.md +++ b/context/project-overview.md @@ -67,8 +67,17 @@ Ghost AI is a real-time collaborative system design workspace. Users describe a ### Spec Generation - The current canvas graph is converted into a Markdown technical specification. -- Specs are persisted as files and linked to the project in the database. -- Users can view and download generated specs. +- Owners and collaborators generate a spec from the **Specs** tab in the AI sidebar. It runs in the background and + downloads automatically as `{project-name}-spec.md` when ready; **Download .md** fetches the latest spec again. +- The document contains an overview and goals, key decisions highlighted as callouts (with rationale, alternatives, + trade-offs, the components involved, and whether each was recorded during AI design, stated on the canvas, or + inferred), one section per group of connected components with a Mermaid diagram, component table, flow and notes, + a component reference, a connections table, and risks and open questions. +- Diagrams, tables, and connections are built from the canvas in code; the model writes only the explanations and + cannot add components or connections. Decisions recorded in the requesting user's AI design sessions are included. +- One spec per project: generating again replaces it. The Markdown is persisted in blob storage and linked to the + project in the database. Spec history is out of scope. +- An empty canvas, or one with more than 200 components, is rejected with a clear message. ## Scope diff --git a/context/ui-context.md b/context/ui-context.md index c6cb15c..c4fee1e 100644 --- a/context/ui-context.md +++ b/context/ui-context.md @@ -12,7 +12,8 @@ Two separate visual systems live in this app and must not be mixed: canvas status panel use the brand tokens too, and so do the project, share and starter-templates dialogs (all built on `PaperDialog` in `components/editor/paper-dialog.tsx`) and the Clerk user menu (`UserMenuButton`). Inside the AI sidebar, the session bar, saved-chat list (`AiSessionHistory`) and chat cards (`QuestionsCard`, - `PlanCard`, `ResultCard` in `components/editor/ai-chat-cards.tsx`) follow the same paper conventions: + `PlanCard`, `ResultCard` in `components/editor/ai-chat-cards.tsx`) and the Specs tab (`SpecPanel` in + `components/editor/spec-panel.tsx`) follow the same paper conventions: `paper-bright` cards with ink hairlines, mono chrome kickers, chips that fill ink when selected, ink primary buttons, and a plan's key decisions on marker-amber paper. Live cursors, the AI thinking indicator and Clerk's "Manage account" profile modal are still product-dark. diff --git a/hooks/use-spec-generator.ts b/hooks/use-spec-generator.ts new file mode 100644 index 0000000..217e635 --- /dev/null +++ b/hooks/use-spec-generator.ts @@ -0,0 +1,236 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import { useRealtimeRun } from "@trigger.dev/react-hooks"; + +import { + parseSpecAgentStage, + SPEC_AGENT_STAGE_KEY, + type SpecAgentStage, + type SpecStatus, + type SpecSummary, +} from "@/lib/spec/spec-schema"; +import type { specAgentTask } from "@/src/trigger/spec-agent"; + +export interface UseSpecGeneratorResult { + /** The project's latest stored spec. */ + spec: SpecSummary | null; + /** True until the first status load finishes. */ + isLoading: boolean; + /** True from clicking generate until the run has finished and been stored. */ + isRunning: boolean; + statusText: string | null; + error: string | null; + /** Same-origin download URL for the stored spec. */ + downloadHref: string; + generate: () => void; +} + +const STAGE_LABELS: Record = { + reading: "Reading the canvas…", + writing: "Writing the spec…", + rendering: "Formatting the Markdown…", + done: "Saving the spec…", +}; + +const STARTING_STATUS = "Starting…"; + +/** While a run is pending the status is re-read on this interval; reading it stores a finished spec. */ +const PENDING_POLL_MS = 4000; + +async function fetchSpecStatus(projectId: string): Promise { + try { + const response = await fetch(`/api/projects/${projectId}/spec`, { cache: "no-store" }); + if (!response.ok) { + return null; + } + return (await response.json()) as SpecStatus; + } catch { + return null; + } +} + +function describeStartFailure(status: number, serverError: string | undefined): string { + switch (status) { + case 401: + return "Your session expired. Sign in again to generate a spec."; + case 403: + return "You do not have access to this project, so no spec was generated."; + case 502: + return "The spec service is unreachable right now. Try again shortly."; + default: + return serverError ?? "The spec could not be started. Try again."; + } +} + +function startDownload(href: string) { + const link = document.createElement("a"); + link.href = href; + link.download = ""; + document.body.appendChild(link); + link.click(); + link.remove(); +} + +/** + * Drives the Specs tab: loads the project's spec status, starts generation, + * follows the run (Realtime for stages, polling as the guarantee), and + * downloads the Markdown automatically once the run the user started is stored. + */ +export function useSpecGenerator(projectId: string): UseSpecGeneratorResult { + const [status, setStatus] = useState(null); + const [isLoading, setIsLoading] = useState(true); + const [isSubmitting, setIsSubmitting] = useState(false); + const [requestError, setRequestError] = useState(null); + const [runAccess, setRunAccess] = useState<{ runId: string; token: string } | null>(null); + + const tokenRequestedFor = useRef(null); + /** The run whose stored spec should download automatically. */ + const downloadWhenStored = useRef(null); + + const downloadHref = `/api/projects/${projectId}/spec/markdown`; + const pendingRunId = status?.pendingRunId ?? null; + + useEffect(() => { + let cancelled = false; + void (async () => { + const next = await fetchSpecStatus(projectId); + if (cancelled) { + return; + } + if (next) { + setStatus(next); + } + setIsLoading(false); + })(); + return () => { + cancelled = true; + }; + }, [projectId]); + + // Exchange the pending run id for a run-scoped Realtime token. + useEffect(() => { + if (!pendingRunId || tokenRequestedFor.current === pendingRunId) { + return; + } + tokenRequestedFor.current = pendingRunId; + + void (async () => { + try { + const response = await fetch("/api/ai/design/token", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ runId: pendingRunId }), + }); + if (!response.ok) { + return; // Only the user who started a run can watch it; polling still stores it. + } + const { token } = (await response.json()) as { token: string }; + setRunAccess({ runId: pendingRunId, token }); + } catch { + // Polling still stores the spec. + } + })(); + }, [pendingRunId]); + + const { run, error: realtimeError } = useRealtimeRun(runAccess?.runId, { + accessToken: runAccess?.token, + enabled: Boolean(runAccess && runAccess.runId === pendingRunId), + }); + + const activeRun = run && run.id === pendingRunId ? run : undefined; + const runFinished = + pendingRunId !== null && + (Boolean(activeRun && (activeRun.isSuccess || activeRun.isFailed || activeRun.isCancelled)) || + Boolean(realtimeError)); + + // Realtime reports the end of a run sooner than the next poll would. + useEffect(() => { + if (!runFinished) { + return; + } + void (async () => { + const next = await fetchSpecStatus(projectId); + if (next) { + setStatus(next); + } + })(); + }, [projectId, runFinished]); + + useEffect(() => { + if (!pendingRunId) { + return; + } + const timer = window.setInterval(() => { + void (async () => { + const next = await fetchSpecStatus(projectId); + if (next) { + setStatus(next); + } + })(); + }, PENDING_POLL_MS); + return () => window.clearInterval(timer); + }, [pendingRunId, projectId]); + + // Download once the run the user started has been stored; give up if it failed. + useEffect(() => { + const awaited = downloadWhenStored.current; + if (!awaited || !status) { + return; + } + if (status.spec?.runId === awaited) { + downloadWhenStored.current = null; + startDownload(downloadHref); + } else if (!status.pendingRunId && status.lastFailure) { + downloadWhenStored.current = null; + } + }, [downloadHref, status]); + + const generate = useCallback(() => { + if (isSubmitting || pendingRunId) { + return; + } + + setRequestError(null); + setIsSubmitting(true); + + void (async () => { + try { + const response = await fetch(`/api/projects/${projectId}/spec`, { method: "POST" }); + const body = (await response.json().catch(() => null)) as { runId?: string; error?: string } | null; + + // 409 means a spec is already being generated; follow that run instead. + if ((response.status === 202 || response.status === 409) && body?.runId) { + const runId = body.runId; + downloadWhenStored.current = runId; + setStatus((previous) => ({ spec: previous?.spec ?? null, pendingRunId: runId, lastFailure: null })); + return; + } + + setRequestError(describeStartFailure(response.status, body?.error)); + } catch { + setRequestError("The spec could not be started. Try again."); + } finally { + setIsSubmitting(false); + } + })(); + }, [isSubmitting, pendingRunId, projectId]); + + const isRunning = isSubmitting || pendingRunId !== null; + + let statusText: string | null = null; + if (isRunning) { + const stage = activeRun ? parseSpecAgentStage(activeRun.metadata?.[SPEC_AGENT_STAGE_KEY]) : null; + statusText = stage ? STAGE_LABELS[stage] : STARTING_STATUS; + } + + return { + spec: status?.spec ?? null, + isLoading, + isRunning, + statusText, + error: requestError ?? (isRunning ? null : (status?.lastFailure ?? null)), + downloadHref, + generate, + }; +} diff --git a/lib/ai/design-agent-engine.ts b/lib/ai/design-agent-engine.ts index 9ccac1a..9107f4a 100644 --- a/lib/ai/design-agent-engine.ts +++ b/lib/ai/design-agent-engine.ts @@ -1,5 +1,4 @@ -import { google, type GoogleLanguageModelOptions } from "@ai-sdk/google"; -import { generateText, NoObjectGeneratedError, Output, type LanguageModel, type ModelMessage } from "ai"; +import { generateText, Output, type LanguageModel, type ModelMessage } from "ai"; import { clampBrief, @@ -16,6 +15,7 @@ import { } from "@/lib/ai/agent-schema"; import { formatCanvasSummary, type CanvasSummary } from "@/lib/ai/canvas-summary"; import { validateDesignGraph } from "@/lib/ai/graph-validation"; +import { thinking, withSchemaRetry } from "@/lib/ai/model"; import { ANALYZE_SYSTEM_PROMPT, buildAnalyzeContext, @@ -30,7 +30,7 @@ import { designGraphSchema, type DesignGraph } from "@/lib/design-generation"; // Model calls for one design turn. No canvas or database access, so the steps // can be exercised directly from a script. -const DEFAULT_MODEL_ID = "gemini-3.5-flash"; +export { resolveDesignModel } from "@/lib/ai/model"; /** * Upper bound for each structured call, retries included. Under provider load a @@ -43,55 +43,6 @@ const CALL_TIMEOUT_MS = { generate: 120_000, } as const; -type ThinkingLevel = "low" | "medium"; - -/** Gemini 2.x models take a token budget instead of a thinking level. */ -const THINKING_BUDGETS: Record = { - low: 1024, - medium: 4096, -}; - -export function resolveDesignModel(): LanguageModel { - const configured = process.env.GOOGLE_GENERATIVE_AI_MODEL?.trim(); - return google(configured && configured.length > 0 ? configured : DEFAULT_MODEL_ID); -} - -function modelIdOf(model: LanguageModel): string { - return typeof model === "string" ? model : model.modelId; -} - -/** - * Model reasoning depth per step. Default thinking made a single turn take up - * to ~80s; reading a turn and drawing an agreed plan need little reasoning, - * while drafting the plan (where the architectural decisions are made) keeps - * more. Gemini 3+ takes `thinkingLevel`; Gemini 2.x only understands - * `thinkingBudget`. - */ -function thinking(model: LanguageModel, level: ThinkingLevel) { - const thinkingConfig = /^gemini-2\./.test(modelIdOf(model)) - ? { thinkingBudget: THINKING_BUDGETS[level] } - : { thinkingLevel: level }; - return { google: { thinkingConfig } satisfies GoogleLanguageModelOptions }; -} - -/** - * Runs a structured-output call, retrying once when the response does not - * match the schema. Transport errors are already retried by the SDK; a schema - * mismatch is usually a one-off bad sample, so a single fresh attempt is - * cheaper than failing the whole turn. - */ -async function withSchemaRetry(call: () => Promise): Promise { - try { - return await call(); - } catch (error) { - if (!NoObjectGeneratedError.isInstance(error)) { - throw error; - } - console.warn("[design-agent] model output did not match the schema; retrying once", error.cause); - return call(); - } -} - /** * Builds the model conversation from stored history plus the new turn. Leading * assistant entries are dropped, since a trimmed history can start mid-reply diff --git a/lib/ai/model.ts b/lib/ai/model.ts new file mode 100644 index 0000000..a15ab7b --- /dev/null +++ b/lib/ai/model.ts @@ -0,0 +1,54 @@ +import { google, type GoogleLanguageModelOptions } from "@ai-sdk/google"; +import { NoObjectGeneratedError, type LanguageModel } from "ai"; + +// Model configuration shared by the Trigger.dev agents (design and spec). + +const DEFAULT_MODEL_ID = "gemini-3.5-flash"; + +export type ThinkingLevel = "low" | "medium"; + +/** Gemini 2.x models take a token budget instead of a thinking level. */ +const THINKING_BUDGETS: Record = { + low: 1024, + medium: 4096, +}; + +export function resolveDesignModel(): LanguageModel { + const configured = process.env.GOOGLE_GENERATIVE_AI_MODEL?.trim(); + return google(configured && configured.length > 0 ? configured : DEFAULT_MODEL_ID); +} + +export function modelIdOf(model: LanguageModel): string { + return typeof model === "string" ? model : model.modelId; +} + +/** + * Model reasoning depth for a step. Default thinking made a single call take up + * to ~80s; most steps need little reasoning, while drafting plans and specs + * keeps more. Gemini 3+ takes `thinkingLevel`; Gemini 2.x only understands + * `thinkingBudget`. + */ +export function thinking(model: LanguageModel, level: ThinkingLevel) { + const thinkingConfig = /^gemini-2\./.test(modelIdOf(model)) + ? { thinkingBudget: THINKING_BUDGETS[level] } + : { thinkingLevel: level }; + return { google: { thinkingConfig } satisfies GoogleLanguageModelOptions }; +} + +/** + * Runs a structured-output call, retrying once when the response does not + * match the schema. Transport errors are already retried by the SDK; a schema + * mismatch is usually a one-off bad sample, so a single fresh attempt is + * cheaper than failing the whole run. + */ +export async function withSchemaRetry(call: () => Promise): Promise { + try { + return await call(); + } catch (error) { + if (!NoObjectGeneratedError.isInstance(error)) { + throw error; + } + console.warn("[ai] model output did not match the schema; retrying once", error.cause); + return call(); + } +} diff --git a/lib/ai/run-failure.ts b/lib/ai/run-failure.ts new file mode 100644 index 0000000..be4a8a4 --- /dev/null +++ b/lib/ai/run-failure.ts @@ -0,0 +1,35 @@ +// Turns a failed Trigger.dev run's error into something a user can act on. +// Raw provider errors (quota details, retry counts, URLs) are logged by the +// caller and never shown. + +interface DescribeRunFailureOptions { + /** Shown when nothing more specific applies. */ + generic: string; + /** Shown when the run timed out. */ + timeout: string; + /** Messages written for users (e.g. an abort reason) that are shown as they are. */ + passthrough?: readonly string[]; +} + +export function describeRunFailure( + message: string | undefined, + { generic, timeout, passthrough = [] }: DescribeRunFailureOptions, +): string { + const trimmed = message?.trim(); + if (!trimmed) { + return generic; + } + if (passthrough.includes(trimmed)) { + return trimmed; + } + if (/quota|rate.?limit|too many requests|\b429\b/i.test(trimmed)) { + return "Draftly AI has hit its usage limit for the moment. Wait a minute and try again."; + } + if (/high demand|overloaded|unavailable|\b503\b/i.test(trimmed)) { + return "Draftly AI is busy right now. Try again in a moment."; + } + if (/timed? ?out|timeout|deadline/i.test(trimmed)) { + return timeout; + } + return generic; +} diff --git a/lib/ai/session-turns.ts b/lib/ai/session-turns.ts index e5cc966..9d1223f 100644 --- a/lib/ai/session-turns.ts +++ b/lib/ai/session-turns.ts @@ -29,6 +29,7 @@ import { MAX_MESSAGES_PER_SESSION, toSessionTitle, } from "@/lib/ai/session-store"; +import { describeRunFailure } from "@/lib/ai/run-failure"; import { prisma } from "@/lib/prisma"; import type { designAgentTask } from "@/src/trigger/design-agent"; import type { @@ -123,26 +124,13 @@ function failedOutcome(content: string): TurnOutcome { return { kind: "ERROR", status: "FAILED", content }; } -/** - * Turns a failed run's error into something a user can act on. Raw provider - * errors (quota details, retry counts, URLs) are logged, never stored. - */ -function describeRunFailure(runId: string, message: string | undefined): string { +/** Logs a failed run's raw error and returns the message stored for the user. */ +function describeFailedRun(runId: string, message: string | undefined): string { console.error("[ai-sessions] design-agent run failed", runId, message); - - if (!message) { - return GENERIC_RUN_ERROR; - } - if (/quota|rate.?limit|too many requests|\b429\b/i.test(message)) { - return "Draftly AI has hit its usage limit for the moment. Wait a minute and try again."; - } - if (/high demand|overloaded|unavailable|\b503\b/i.test(message)) { - return "Draftly AI is busy right now. Try again in a moment."; - } - if (/timed? ?out|timeout|deadline/i.test(message)) { - return "The design agent took too long to respond. Try again."; - } - return GENERIC_RUN_ERROR; + return describeRunFailure(message, { + generic: GENERIC_RUN_ERROR, + timeout: "The design agent took too long to respond. Try again.", + }); } function outcomeFromResult(result: DesignAgentResult): TurnOutcome { @@ -243,7 +231,7 @@ async function settleTurn(sessionId: string, messageId: string, runId: string): } else if (run.isCancelled) { outcome = failedOutcome("The design run was cancelled."); } else if (run.isFailed) { - outcome = failedOutcome(describeRunFailure(runId, run.error?.message)); + outcome = failedOutcome(describeFailedRun(runId, run.error?.message)); } else { return false; } diff --git a/lib/canvas-room.ts b/lib/canvas-room.ts new file mode 100644 index 0000000..78e4149 --- /dev/null +++ b/lib/canvas-room.ts @@ -0,0 +1,15 @@ +import { mutateFlow } from "@liveblocks/react-flow/node"; + +import { getLiveblocksClient } from "@/lib/liveblocks"; +import type { CanvasEdge, CanvasNode } from "@/types/canvas"; + +/** A copy of a room's nodes and edges, read without changing the room. */ +export async function readRoomSnapshot(roomId: string): Promise<{ nodes: CanvasNode[]; edges: CanvasEdge[] }> { + let snapshot: { nodes: CanvasNode[]; edges: CanvasEdge[] } = { nodes: [], edges: [] }; + + await mutateFlow({ client: getLiveblocksClient(), roomId }, (flow) => { + snapshot = { nodes: [...flow.nodes], edges: [...flow.edges] }; + }); + + return snapshot; +} diff --git a/lib/spec/mermaid.ts b/lib/spec/mermaid.ts new file mode 100644 index 0000000..0c39280 --- /dev/null +++ b/lib/spec/mermaid.ts @@ -0,0 +1,69 @@ +import type { SpecGraph, SpecGroup } from "@/lib/spec/spec-graph"; +import type { CanvasArrowDirection, CanvasShape } from "@/types/canvas"; + +// Mermaid flowcharts for spec diagram sections. Every node label is quoted and +// entity-escaped, so component names cannot break the diagram syntax. + +const SHAPE_WRAPPERS: Record = { + rectangle: ['["', '"]'], + circle: ['(("', '"))'], + diamond: ['{"', '"}'], + pill: ['(["', '"])'], + cylinder: ['[("', '")]'], + hexagon: ['{{"', '"}}'], +}; + +/** Text safe inside a quoted Mermaid label. */ +export function escapeMermaidText(text: string): string { + return text + .replace(/\s+/g, " ") + .trim() + .replace(/#/g, "#35;") + .replace(/"/g, "#quot;") + .replace(//g, "#gt;") + .replace(/\|/g, "#124;"); +} + +function edgeOperator(direction: CanvasArrowDirection, async: boolean): { operator: string; reverse: boolean } { + switch (direction) { + case "forward": + return { operator: async ? "-.->" : "-->", reverse: false }; + case "backward": + return { operator: async ? "-.->" : "-->", reverse: true }; + case "bidirectional": + return { operator: async ? "<-.->" : "<-->", reverse: false }; + default: + return { operator: async ? "-.-" : "---", reverse: false }; + } +} + +/** The group as a top-down Mermaid flowchart, without code fences. */ +export function renderGroupMermaid(graph: SpecGraph, group: SpecGroup): string { + const componentByRef = new Map(graph.components.map((component) => [component.ref, component])); + const lines = ["flowchart TD"]; + + for (const ref of group.componentRefs) { + const component = componentByRef.get(ref); + if (!component) { + continue; + } + const [open, close] = SHAPE_WRAPPERS[component.shape]; + lines.push(` ${ref}${open}${escapeMermaidText(component.label)}${close}`); + } + + for (const index of group.connectionIndexes) { + const connection = graph.connections[index]; + if (!connection) { + continue; + } + const { operator, reverse } = edgeOperator(connection.direction, connection.async); + const [from, to] = reverse + ? [connection.targetRef, connection.sourceRef] + : [connection.sourceRef, connection.targetRef]; + const label = connection.label ? `|"${escapeMermaidText(connection.label)}"|` : ""; + lines.push(` ${from} ${operator}${label} ${to}`); + } + + return lines.join("\n"); +} diff --git a/lib/spec/render-markdown.ts b/lib/spec/render-markdown.ts new file mode 100644 index 0000000..353759f --- /dev/null +++ b/lib/spec/render-markdown.ts @@ -0,0 +1,229 @@ +import { renderGroupMermaid } from "@/lib/spec/mermaid"; +import type { SpecConnection, SpecGraph, SpecGroup } from "@/lib/spec/spec-graph"; +import type { DecisionSource, SpecContent, SpecStats } from "@/lib/spec/spec-schema"; + +// --------------------------------------------------------------------------- +// Markdown rendering +// +// The spec document is assembled in code: headings, callouts, Mermaid +// diagrams, and tables come from the canvas graph, and the model's prose fills +// in the explanations. GitHub renders the `> [!IMPORTANT]` callouts and +// Mermaid blocks natively. +// --------------------------------------------------------------------------- + +export interface RenderSpecOptions { + projectName: string; + generatedAt: Date; + modelId: string; +} + +const SOURCE_LABELS: Record = { + recorded: "Recorded during AI design", + stated: "Stated on the canvas", + inferred: "Inferred from the diagram", +}; + +function inline(text: string): string { + return text.replace(/\s+/g, " ").trim(); +} + +/** Text safe inside a Markdown table cell. */ +function cell(text: string | undefined): string { + const value = inline(text ?? "").replace(/\|/g, "\\|"); + return value.length > 0 ? value : "—"; +} + +function plural(count: number, noun: string): string { + return `${count} ${noun}${count === 1 ? "" : "s"}`; +} + +function formatUtc(date: Date): string { + return `${date.toISOString().slice(0, 16).replace("T", " ")} UTC`; +} + +export function renderSpecMarkdown( + graph: SpecGraph, + content: SpecContent, + { projectName, generatedAt, modelId }: RenderSpecOptions, +): { markdown: string; stats: SpecStats } { + const componentByRef = new Map(graph.components.map((component) => [component.ref, component])); + const groupContent = new Map(content.groups.map((group) => [group.groupId, group])); + const responsibilities = new Map(content.components.map((entry) => [entry.componentId, entry.responsibility])); + const nameOf = (ref: string) => componentByRef.get(ref)?.label ?? ref; + const groupName = (group: SpecGroup, index: number) => + group.standalone ? "Standalone components" : groupContent.get(group.id)?.name || `Diagram ${index + 1}`; + const groupNameById = new Map(graph.groups.map((group, index) => [group.id, groupName(group, index)])); + + const stats: SpecStats = { + components: graph.components.length, + connections: graph.connections.length, + diagrams: graph.groups.length, + decisions: content.decisions.length, + }; + + const title = inline(projectName) || "Untitled project"; + const out: string[] = [ + `# ${title} — Technical Specification`, + "", + `> Generated by Draftly on ${formatUtc(generatedAt)} from the project canvas: ${plural(stats.components, "component")}, ` + + `${plural(stats.connections, "connection")}, ${plural(stats.diagrams, "diagram")}.`, + "", + "## Contents", + "", + "1. [Overview](#1-overview)", + "2. [Key decisions](#2-key-decisions)", + "3. [System diagrams](#3-system-diagrams)", + "4. [Component reference](#4-component-reference)", + "5. [Connections](#5-connections)", + "6. [Risks and open questions](#6-risks-and-open-questions)", + "", + ]; + + // ---- 1. Overview + out.push("## 1. Overview", "", content.overview || "_No overview was produced._", ""); + if (content.goals.length > 0) { + out.push("**Goals**", "", ...content.goals.map((goal) => `- ${inline(goal)}`), ""); + } + if (graph.generalNotes.length > 0) { + out.push("**Notes on the canvas**", "", ...graph.generalNotes.map((note) => `- ${inline(note)}`), ""); + } + + // ---- 2. Key decisions + out.push("## 2. Key decisions", ""); + if (content.decisions.length === 0) { + out.push("_No major decisions were identified from the canvas._", ""); + } + for (const decision of content.decisions) { + const paragraphs = [ + `**${inline(decision.title)}:** ${inline(decision.decision)}`, + decision.rationale ? `**Why:** ${inline(decision.rationale)}` : "", + decision.alternatives.length > 0 ? `**Alternatives considered:** ${decision.alternatives.map(inline).join("; ")}` : "", + decision.tradeoffs ? `**Trade-offs:** ${inline(decision.tradeoffs)}` : "", + [ + decision.componentIds.length > 0 ? `**Components:** ${decision.componentIds.map(nameOf).join(", ")}` : "", + `**Source:** ${SOURCE_LABELS[decision.source]}`, + ] + .filter(Boolean) + .join(" · "), + ].filter(Boolean); + + out.push("> [!IMPORTANT]", ...paragraphs.flatMap((paragraph, index) => (index === 0 ? [`> ${paragraph}`] : [">", `> ${paragraph}`])), ""); + } + + // ---- 3. System diagrams + out.push("## 3. System diagrams", ""); + if (graph.groups.length === 0) { + out.push("_The canvas has no components._", ""); + } + graph.groups.forEach((group, index) => { + const described = groupContent.get(group.id); + out.push(`### 3.${index + 1} ${groupName(group, index)}`, ""); + if (described?.purpose) { + out.push(described.purpose, ""); + } + out.push("```mermaid", renderGroupMermaid(graph, group), "```", ""); + out.push( + "| Component | Kind | Responsibility |", + "| --- | --- | --- |", + ...group.componentRefs.map((ref) => { + const component = componentByRef.get(ref); + return `| ${cell(component?.label)} | ${cell(component?.kind)} | ${cell(responsibilities.get(ref))} |`; + }), + "", + ); + if (described && described.flow.length > 0) { + out.push("**Flow**", "", ...described.flow.map((step, stepIndex) => `${stepIndex + 1}. ${inline(step)}`), ""); + } + if (group.notes.length > 0) { + out.push("**Notes**", "", ...group.notes.map((note) => `- ${inline(note)}`), ""); + } + }); + + // ---- 4. Component reference + out.push("## 4. Component reference", ""); + if (graph.components.length === 0) { + out.push("_The canvas has no components._", ""); + } + for (const component of graph.components) { + const receivesFrom: string[] = []; + const sendsTo: string[] = []; + const connectedTo: string[] = []; + + for (const connection of graph.connections) { + if (connection.sourceRef !== component.ref && connection.targetRef !== component.ref) { + continue; + } + const isSource = connection.sourceRef === component.ref; + const other = nameOf(isSource ? connection.targetRef : connection.sourceRef); + const details = [connection.label, connection.async ? "async" : ""].filter(Boolean).join(", "); + const entry = details ? `${other} (${inline(details)})` : other; + + switch (connection.direction) { + case "forward": + (isSource ? sendsTo : receivesFrom).push(entry); + break; + case "backward": + (isSource ? receivesFrom : sendsTo).push(entry); + break; + case "bidirectional": + sendsTo.push(entry); + receivesFrom.push(entry); + break; + default: + connectedTo.push(entry); + } + } + + out.push( + `### ${component.label}`, + "", + `- **Kind:** ${inline(component.kind)}`, + `- **Diagram:** ${groupNameById.get(component.groupId) ?? "—"}`, + `- **Responsibility:** ${responsibilities.get(component.ref) ?? "_Not described._"}`, + ...(receivesFrom.length > 0 ? [`- **Receives from:** ${receivesFrom.join(", ")}`] : []), + ...(sendsTo.length > 0 ? [`- **Sends to:** ${sendsTo.join(", ")}`] : []), + ...(connectedTo.length > 0 ? [`- **Connected to:** ${connectedTo.join(", ")}`] : []), + "", + ); + } + + // ---- 5. Connections + out.push("## 5. Connections", ""); + if (graph.connections.length === 0) { + out.push("_The canvas has no connections._", ""); + } else { + const describe = (connection: SpecConnection) => { + const [from, to] = + connection.direction === "backward" + ? [connection.targetRef, connection.sourceRef] + : [connection.sourceRef, connection.targetRef]; + const direction = + connection.direction === "bidirectional" ? "Two-way" : connection.direction === "none" ? "Undirected" : "One-way"; + return `| ${cell(nameOf(from))} | ${cell(nameOf(to))} | ${cell(connection.label)} | ${connection.async ? "Async" : "Sync"} | ${direction} |`; + }; + out.push("| From | To | Label | Delivery | Direction |", "| --- | --- | --- | --- | --- |", ...graph.connections.map(describe), ""); + } + + // ---- 6. Risks and open questions + out.push("## 6. Risks and open questions", "", "**Risks**", ""); + if (content.risks.length === 0) { + out.push("_No risks were identified._", ""); + } else { + out.push("| Risk | Mitigation |", "| --- | --- |", ...content.risks.map((risk) => `| ${cell(risk.risk)} | ${cell(risk.mitigation)} |`), ""); + } + out.push("**Open questions**", ""); + if (content.openQuestions.length === 0) { + out.push("_None recorded._", ""); + } else { + out.push(...content.openQuestions.map((question) => `- [ ] ${inline(question)}`), ""); + } + + out.push( + "---", + "", + `_Generated with Draftly using ${modelId} on ${formatUtc(generatedAt)}. Diagrams reflect the canvas at generation time._`, + ); + + const markdown = `${out.join("\n").replace(/\n{3,}/g, "\n\n").trimEnd()}\n`; + return { markdown, stats }; +} diff --git a/lib/spec/spec-engine.ts b/lib/spec/spec-engine.ts new file mode 100644 index 0000000..9e35c40 --- /dev/null +++ b/lib/spec/spec-engine.ts @@ -0,0 +1,38 @@ +import { generateText, Output, type LanguageModel } from "ai"; + +import { thinking, withSchemaRetry } from "@/lib/ai/model"; +import type { SpecGraph } from "@/lib/spec/spec-graph"; +import { buildSpecPrompt, SPEC_SYSTEM_PROMPT } from "@/lib/spec/spec-prompt"; +import { + sanitizeSpecContent, + specContentSchema, + type RecordedDecision, + type SpecContent, +} from "@/lib/spec/spec-schema"; + +/** Upper bound for the spec call, retries included. */ +const SPEC_CALL_TIMEOUT_MS = 180_000; + +/** Asks the model for the spec's prose and drops anything that doesn't match the canvas. */ +export async function generateSpecContent( + model: LanguageModel, + graph: SpecGraph, + projectName: string, + recordedDecisions: readonly RecordedDecision[], +): Promise { + const { output } = await withSchemaRetry(() => + generateText({ + model, + system: SPEC_SYSTEM_PROMPT, + prompt: buildSpecPrompt(graph, projectName, recordedDecisions), + output: Output.object({ schema: specContentSchema, name: "technical_spec" }), + providerOptions: thinking(model, "medium"), + timeout: { totalMs: SPEC_CALL_TIMEOUT_MS }, + }), + ); + + return sanitizeSpecContent(output, { + groupIds: new Set(graph.groups.map((group) => group.id)), + componentRefs: new Set(graph.components.map((component) => component.ref)), + }); +} diff --git a/lib/spec/spec-file-name.ts b/lib/spec/spec-file-name.ts new file mode 100644 index 0000000..6076f40 --- /dev/null +++ b/lib/spec/spec-file-name.ts @@ -0,0 +1,11 @@ +/** Download file name for a project's spec: `payments-service-spec.md` from "Payments Service". */ +export function toSpecFileName(projectName: string): string { + const slug = projectName + .normalize("NFKD") + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, "") + .slice(0, 60) + .replace(/-+$/g, ""); + return `${slug || "draftly"}-spec.md`; +} diff --git a/lib/spec/spec-graph.ts b/lib/spec/spec-graph.ts new file mode 100644 index 0000000..296ab60 --- /dev/null +++ b/lib/spec/spec-graph.ts @@ -0,0 +1,222 @@ +import { + CANVAS_SHAPES, + SHAPE_DEFAULTS, + SHAPE_KICKERS, + TEXT_NODE_SHAPE, + type CanvasArrowDirection, + type CanvasEdge, + type CanvasNode, + type CanvasShape, +} from "@/types/canvas"; + +// --------------------------------------------------------------------------- +// Spec graph +// +// The canvas as the spec describes it: drawn components split into groups of +// connected components (each group becomes one diagram section), the +// connections between them, and text notes placed with the group they sit +// next to. Components get short refs ("c1") and groups short ids ("g1") that +// the model uses; the renderer maps them back to names. +// --------------------------------------------------------------------------- + +/** Largest canvas a single spec run covers. */ +export const SPEC_MAX_COMPONENTS = 200; + +/** A note further than this from every group is listed as a general note. */ +const NOTE_ATTACH_DISTANCE = 600; + +export interface SpecComponent { + ref: string; + nodeId: string; + label: string; + kind: string; + shape: CanvasShape; + groupId: string; +} + +export interface SpecConnection { + sourceRef: string; + targetRef: string; + label?: string; + async: boolean; + direction: CanvasArrowDirection; +} + +export interface SpecGroup { + /** "g1", "g2", … in reading order, or "standalone". */ + id: string; + standalone: boolean; + componentRefs: string[]; + /** Indexes into `SpecGraph.connections`. */ + connectionIndexes: number[]; + notes: string[]; +} + +export interface SpecGraph { + components: SpecComponent[]; + connections: SpecConnection[]; + groups: SpecGroup[]; + /** Notes that are not near any group. */ + generalNotes: string[]; + /** Drawn components on the canvas, including any beyond `SPEC_MAX_COMPONENTS`. */ + totalComponents: number; +} + +interface Box { + x: number; + y: number; + width: number; + height: number; +} + +function labelOf(node: CanvasNode): string { + return typeof node.data?.label === "string" ? node.data.label.trim() : ""; +} + +function resolveShape(shape: unknown): CanvasShape { + return (CANVAS_SHAPES as readonly unknown[]).includes(shape) ? (shape as CanvasShape) : "rectangle"; +} + +function boxOf(node: CanvasNode): Box { + const fallback = node.data?.shape === TEXT_NODE_SHAPE ? SHAPE_DEFAULTS.text : SHAPE_DEFAULTS[resolveShape(node.data?.shape)]; + const width = typeof node.style?.width === "number" ? node.style.width : fallback.width; + const height = typeof node.style?.height === "number" ? node.style.height : fallback.height; + return { x: node.position.x, y: node.position.y, width, height }; +} + +/** Distance from a point to the nearest edge of a box; 0 inside it. */ +function distanceToBox(point: { x: number; y: number }, box: Box): number { + const dx = Math.max(box.x - point.x, 0, point.x - (box.x + box.width)); + const dy = Math.max(box.y - point.y, 0, point.y - (box.y + box.height)); + return Math.hypot(dx, dy); +} + +function boundingBox(boxes: readonly Box[]): Box { + const left = Math.min(...boxes.map((box) => box.x)); + const top = Math.min(...boxes.map((box) => box.y)); + const right = Math.max(...boxes.map((box) => box.x + box.width)); + const bottom = Math.max(...boxes.map((box) => box.y + box.height)); + return { x: left, y: top, width: right - left, height: bottom - top }; +} + +export function buildSpecGraph(nodes: readonly CanvasNode[], edges: readonly CanvasEdge[]): SpecGraph { + const drawnAll = nodes.filter((node) => node.data?.shape !== TEXT_NODE_SHAPE && labelOf(node).length > 0); + const drawn = drawnAll.slice(0, SPEC_MAX_COMPONENTS); + const nodeById = new Map(drawn.map((node) => [node.id, node])); + const boxes = new Map(drawn.map((node) => [node.id, boxOf(node)])); + + const keptEdges = edges.filter( + (edge) => nodeById.has(edge.source) && nodeById.has(edge.target) && edge.source !== edge.target, + ); + + // Union-find over connections: each connected set of components is one diagram. + const parent = new Map(drawn.map((node) => [node.id, node.id])); + const find = (id: string): string => { + let root = id; + while (parent.get(root) !== root) { + root = parent.get(root) as string; + } + let current = id; + while (parent.get(current) !== root) { + const next = parent.get(current) as string; + parent.set(current, root); + current = next; + } + return root; + }; + for (const edge of keptEdges) { + const a = find(edge.source); + const b = find(edge.target); + if (a !== b) { + parent.set(a, b); + } + } + + const clusters = new Map(); + for (const node of drawn) { + const root = find(node.id); + clusters.set(root, [...(clusters.get(root) ?? []), node.id]); + } + + const byPosition = (a: string, b: string) => { + const boxA = boxes.get(a) as Box; + const boxB = boxes.get(b) as Box; + return boxA.y - boxB.y || boxA.x - boxB.x; + }; + const clusterBox = (ids: readonly string[]) => boundingBox(ids.map((id) => boxes.get(id) as Box)); + + const connected = [...clusters.values()] + .filter((ids) => ids.length > 1) + .sort((a, b) => { + const boxA = clusterBox(a); + const boxB = clusterBox(b); + return boxA.y - boxB.y || boxA.x - boxB.x; + }); + const singles = [...clusters.values()].filter((ids) => ids.length === 1).flat(); + + const ordered = connected.map((ids, index) => ({ id: `g${index + 1}`, standalone: false, nodeIds: [...ids].sort(byPosition) })); + if (singles.length > 0) { + ordered.push({ id: "standalone", standalone: true, nodeIds: [...singles].sort(byPosition) }); + } + + const refByNodeId = new Map(); + const components: SpecComponent[] = []; + for (const group of ordered) { + for (const nodeId of group.nodeIds) { + const node = nodeById.get(nodeId) as CanvasNode; + const ref = `c${components.length + 1}`; + const shape = resolveShape(node.data?.shape); + refByNodeId.set(nodeId, ref); + components.push({ + ref, + nodeId, + label: labelOf(node), + kind: node.data?.kicker?.trim() || SHAPE_KICKERS[shape], + shape, + groupId: group.id, + }); + } + } + + const connections: SpecConnection[] = keptEdges.map((edge) => { + const label = edge.data?.label?.trim(); + return { + sourceRef: refByNodeId.get(edge.source) as string, + targetRef: refByNodeId.get(edge.target) as string, + ...(label ? { label } : {}), + async: edge.data?.edgeStyle === "dashed" || edge.data?.edgeStyle === "dotted", + direction: edge.data?.arrowDirection ?? "none", + }; + }); + + const groups: SpecGroup[] = ordered.map((group) => ({ + id: group.id, + standalone: group.standalone, + componentRefs: group.nodeIds.map((nodeId) => refByNodeId.get(nodeId) as string), + connectionIndexes: [], + notes: [], + })); + const groupById = new Map(groups.map((group) => [group.id, group])); + const groupIdByRef = new Map(components.map((component) => [component.ref, component.groupId])); + connections.forEach((connection, index) => { + groupById.get(groupIdByRef.get(connection.sourceRef) as string)?.connectionIndexes.push(index); + }); + + const generalNotes: string[] = []; + const groupBoxes = ordered.map((group) => ({ id: group.id, box: clusterBox(group.nodeIds) })); + for (const note of nodes.filter((node) => node.data?.shape === TEXT_NODE_SHAPE && labelOf(node).length > 0)) { + const box = boxOf(note); + const center = { x: box.x + box.width / 2, y: box.y + box.height / 2 }; + const nearest = groupBoxes + .map((entry) => ({ id: entry.id, distance: distanceToBox(center, entry.box) })) + .sort((a, b) => a.distance - b.distance)[0]; + + if (nearest && nearest.distance <= NOTE_ATTACH_DISTANCE) { + groupById.get(nearest.id)?.notes.push(labelOf(note)); + } else { + generalNotes.push(labelOf(note)); + } + } + + return { components, connections, groups, generalNotes, totalComponents: drawnAll.length }; +} diff --git a/lib/spec/spec-prompt.ts b/lib/spec/spec-prompt.ts new file mode 100644 index 0000000..627f496 --- /dev/null +++ b/lib/spec/spec-prompt.ts @@ -0,0 +1,98 @@ +import type { SpecGraph } from "@/lib/spec/spec-graph"; +import type { RecordedDecision } from "@/lib/spec/spec-schema"; + +export const SPEC_SYSTEM_PROMPT = [ + "You are a staff engineer writing the technical specification for a system design drawn on a", + "collaborative canvas. The design is given as structured data: groups of connected components (each", + "group is one diagram), components with refs, names and kinds, the connections between them (with", + "labels, direction, and whether they are asynchronous), notes the designers wrote on the canvas, and", + "decisions recorded when the design was generated with the AI architect.", + "", + "Explain the design as drawn. Never invent components or connections that are not in the input.", + "- overview: two or three short paragraphs for an engineer new to the system.", + "- goals: what the design is built to achieve, as evident from the canvas.", + "- groups: for every group, a short name, its purpose, and its main flow as ordered steps that follow", + " its actual connections, naming components exactly.", + "- components: a one-sentence responsibility for every component.", + "- decisions: the major architectural decisions the design shows — datastores, sync vs async messaging,", + " caching, gateways and entry points, authentication, service boundaries, scaling, integration", + " boundaries. Identify them from the diagram itself (usually 3 to 6 for a design with several", + " components), and add every recorded decision that still matches the canvas on top of those; recorded", + " decisions are never the whole list. For each, what the design does, why it fits, realistic", + " alternatives, trade-offs, the refs of the components involved, and its source. Most consequential first.", + "- risks: real weaknesses of this design (single points of failure, consistency gaps, scaling limits,", + " security exposure), each with a mitigation.", + "- openQuestions: what a reviewer should resolve before building.", + "Use the group ids and component refs exactly as given. Write plain text in every field: no markdown", + "headings, tables, or code blocks.", +].join("\n"); + +const DIRECTION_ARROWS = { + forward: "→", + backward: "←", + bidirectional: "↔", + none: "—", +} as const; + +/** The canvas and recorded decisions as prompt text. */ +export function buildSpecPrompt( + graph: SpecGraph, + projectName: string, + recordedDecisions: readonly RecordedDecision[], +): string { + const componentByRef = new Map(graph.components.map((component) => [component.ref, component])); + const name = (ref: string) => componentByRef.get(ref)?.label ?? ref; + + const sections = [ + `Project: ${projectName.trim() || "Untitled project"}`, + `Totals: ${graph.components.length} components, ${graph.connections.length} connections, ${graph.groups.length} groups.`, + ]; + + for (const group of graph.groups) { + const lines = [ + group.standalone + ? `Group ${group.id} (components not connected to anything):` + : `Group ${group.id}:`, + " Components:", + ...group.componentRefs.map((ref) => { + const component = componentByRef.get(ref); + return ` - ${ref}: ${component?.label ?? ref} (${component?.kind ?? "Component"})`; + }), + ]; + + if (group.connectionIndexes.length > 0) { + lines.push(" Connections:"); + for (const index of group.connectionIndexes) { + const connection = graph.connections[index]; + const details = [connection.label, connection.async ? "async" : "sync"].filter(Boolean).join(", "); + lines.push( + ` - ${name(connection.sourceRef)} (${connection.sourceRef}) ${DIRECTION_ARROWS[connection.direction]} ` + + `${name(connection.targetRef)} (${connection.targetRef}): ${details}`, + ); + } + } + + if (group.notes.length > 0) { + lines.push(" Notes:", ...group.notes.map((note) => ` - ${note}`)); + } + + sections.push(lines.join("\n")); + } + + if (graph.generalNotes.length > 0) { + sections.push(`Other notes on the canvas:\n${graph.generalNotes.map((note) => `- ${note}`).join("\n")}`); + } + + sections.push( + recordedDecisions.length > 0 + ? `Decisions recorded during AI design sessions:\n${recordedDecisions + .map((decision) => { + const alternatives = decision.alternatives.length > 0 ? ` (alternatives: ${decision.alternatives.join(", ")})` : ""; + return `- ${decision.title}: ${decision.choice} — ${decision.rationale}${alternatives}`; + }) + .join("\n")}` + : "Decisions recorded during AI design sessions: none.", + ); + + return sections.join("\n\n"); +} diff --git a/lib/spec/spec-schema.ts b/lib/spec/spec-schema.ts new file mode 100644 index 0000000..347edac --- /dev/null +++ b/lib/spec/spec-schema.ts @@ -0,0 +1,247 @@ +import { z } from "zod"; + +import { SPEC_MAX_COMPONENTS } from "@/lib/spec/spec-graph"; + +// --------------------------------------------------------------------------- +// Spec generation contract +// +// The model writes only prose about the design (overview, group purposes and +// flows, component responsibilities, decisions, risks). Structure — diagrams, +// tables, connections — comes from the canvas in code. Shared by the task, +// the server (validating run output), and the sidebar (status and stages). +// --------------------------------------------------------------------------- + +/** Metadata key carrying the current {@link SpecAgentStage}. */ +export const SPEC_AGENT_STAGE_KEY = "stage"; + +export const SPEC_AGENT_STAGES = ["reading", "writing", "rendering", "done"] as const; + +export type SpecAgentStage = (typeof SPEC_AGENT_STAGES)[number]; + +export function parseSpecAgentStage(value: unknown): SpecAgentStage | null { + return SPEC_AGENT_STAGES.find((stage) => stage === value) ?? null; +} + +/** Abort reasons written for users; shown as they are. */ +export const SPEC_EMPTY_CANVAS_MESSAGE = "Add some components to the canvas before generating a spec."; +export const SPEC_TOO_LARGE_MESSAGE = `The canvas has more than ${SPEC_MAX_COMPONENTS} components, which is too many for one spec.`; + +/** Size limits for model-written lists; stated in descriptions, trimmed in code. */ +export const SPEC_LIST_LIMITS = { + goals: 6, + flowSteps: 8, + decisions: 8, + alternatives: 3, + risks: 8, + openQuestions: 8, +} as const; + +/** Decisions recorded in AI design sessions passed to one spec run. */ +export const RECORDED_DECISION_LIMIT = 12; + +export const DECISION_SOURCES = ["recorded", "stated", "inferred"] as const; + +export type DecisionSource = (typeof DECISION_SOURCES)[number]; + +function atMost(limit: number, description: string): string { + return `${description} At most ${limit} items.`; +} + +export const specContentSchema = z.object({ + overview: z + .string() + .min(1) + .describe("Two or three short paragraphs: what the system does, how it is structured, and what matters most."), + goals: z + .array(z.string()) + .describe(atMost(SPEC_LIST_LIMITS.goals, "What the design is built to achieve, as evident from the canvas.")), + groups: z + .array( + z.object({ + groupId: z.string().describe("A group id exactly as given in the input, e.g. 'g1' or 'standalone'."), + name: z.string().min(1).describe("A short name for this diagram, e.g. 'Checkout request path'."), + purpose: z.string().describe("One or two sentences on what this part of the system does."), + flow: z + .array(z.string()) + .describe( + atMost( + SPEC_LIST_LIMITS.flowSteps, + "The main flow through this group as ordered steps that follow its actual connections, naming components exactly.", + ), + ), + }), + ) + .describe("One entry per group in the input."), + components: z + .array( + z.object({ + componentId: z.string().describe("A component ref exactly as given in the input, e.g. 'c3'."), + responsibility: z.string().describe("What the component is responsible for, in one sentence."), + }), + ) + .describe("One entry per component in the input."), + decisions: z + .array( + z.object({ + title: z.string().min(1).describe("The question the decision answers, e.g. 'Primary datastore'."), + decision: z.string().min(1).describe("What the design does."), + rationale: z.string().describe("Why it fits this system."), + alternatives: z.array(z.string()).describe(atMost(SPEC_LIST_LIMITS.alternatives, "Realistic alternatives.")), + tradeoffs: z.string().describe("What the choice costs or risks."), + componentIds: z.array(z.string()).describe("Refs of the components involved."), + source: z + .enum(DECISION_SOURCES) + .describe( + "'recorded' when it comes from the recorded decisions, 'stated' when a canvas note or connection " + + "label states it, otherwise 'inferred'.", + ), + }), + ) + .describe(atMost(SPEC_LIST_LIMITS.decisions, "The major architectural decisions, most consequential first.")), + risks: z + .array(z.object({ risk: z.string(), mitigation: z.string() })) + .describe(atMost(SPEC_LIST_LIMITS.risks, "Risks in the design, each with a mitigation.")), + openQuestions: z + .array(z.string()) + .describe(atMost(SPEC_LIST_LIMITS.openQuestions, "Questions a reviewer should resolve.")), +}); + +export type SpecContent = z.infer; + +export interface KnownSpecIds { + groupIds: ReadonlySet; + componentRefs: ReadonlySet; +} + +/** Parenthesised refs such as "(c3)", "(c1, c4)", "(g2)" or "(standalone)" copied into prose. */ +const SPEC_REF_MENTION = /[ \t]*\((?:\s*(?:c\d+|g\d+|standalone)\s*,?)+\)/g; + +/** Removes internal refs from text shown to readers, who only know components by name. */ +export function withoutSpecRefs(text: string): string { + return text.replace(SPEC_REF_MENTION, ""); +} + +/** + * Trims the model's prose to the list limits, removes internal refs the model + * copied into the text, and drops anything that refers to a group or component + * the canvas does not have. + */ +export function sanitizeSpecContent(content: SpecContent, known: KnownSpecIds): SpecContent { + const clean = (value: string) => withoutSpecRefs(value).replace(/\s+/g, " ").trim(); + const cleanList = (values: readonly string[], limit: number) => values.map(clean).filter(Boolean).slice(0, limit); + + const seenGroups = new Set(); + const groups = content.groups.flatMap((group) => { + if (!known.groupIds.has(group.groupId) || seenGroups.has(group.groupId)) { + return []; + } + seenGroups.add(group.groupId); + return [ + { + groupId: group.groupId, + name: clean(group.name), + purpose: clean(group.purpose), + flow: cleanList(group.flow, SPEC_LIST_LIMITS.flowSteps), + }, + ]; + }); + + const seenComponents = new Set(); + const components = content.components.flatMap((component) => { + const responsibility = clean(component.responsibility); + if (!known.componentRefs.has(component.componentId) || seenComponents.has(component.componentId) || !responsibility) { + return []; + } + seenComponents.add(component.componentId); + return [{ componentId: component.componentId, responsibility }]; + }); + + const decisions = content.decisions + .flatMap((decision) => { + const title = clean(decision.title); + const choice = clean(decision.decision); + if (!title || !choice) { + return []; + } + return [ + { + title, + decision: choice, + rationale: clean(decision.rationale), + alternatives: cleanList(decision.alternatives, SPEC_LIST_LIMITS.alternatives), + tradeoffs: clean(decision.tradeoffs), + componentIds: [...new Set(decision.componentIds.filter((id) => known.componentRefs.has(id)))], + source: decision.source, + }, + ]; + }) + .slice(0, SPEC_LIST_LIMITS.decisions); + + const risks = content.risks + .flatMap((entry) => { + const risk = clean(entry.risk); + return risk ? [{ risk, mitigation: clean(entry.mitigation) }] : []; + }) + .slice(0, SPEC_LIST_LIMITS.risks); + + return { + overview: withoutSpecRefs(content.overview).trim(), + goals: cleanList(content.goals, SPEC_LIST_LIMITS.goals), + groups, + components, + decisions, + risks, + openQuestions: cleanList(content.openQuestions, SPEC_LIST_LIMITS.openQuestions), + }; +} + +// --------------------------------------------------------------------------- +// Run contract and status +// --------------------------------------------------------------------------- + +export interface RecordedDecision { + title: string; + choice: string; + rationale: string; + alternatives: string[]; +} + +export interface SpecAgentPayload { + roomId: string; + projectName: string; + /** Decisions stored in the requesting user's AI design sessions, newest first. */ + recordedDecisions: RecordedDecision[]; +} + +export const specStatsSchema = z.object({ + components: z.number().int().nonnegative(), + connections: z.number().int().nonnegative(), + diagrams: z.number().int().nonnegative(), + decisions: z.number().int().nonnegative(), +}); + +export type SpecStats = z.infer; + +/** Output of one spec-agent run. The server validates it before storing the Markdown. */ +export const specAgentResultSchema = z.object({ + markdown: z.string().min(1), + stats: specStatsSchema, +}); + +export type SpecAgentResult = z.infer; + +export interface SpecSummary { + runId: string; + generatedAt: string; + stats: SpecStats | null; +} + +/** What `GET /api/projects/[projectId]/spec` returns. */ +export interface SpecStatus { + /** The stored spec, if one has been generated. */ + spec: SpecSummary | null; + /** A spec run that is still in progress. */ + pendingRunId: string | null; + /** Why the latest run failed, when it is newer than the stored spec. */ + lastFailure: string | null; +} diff --git a/lib/spec/spec-store.ts b/lib/spec/spec-store.ts new file mode 100644 index 0000000..5629977 --- /dev/null +++ b/lib/spec/spec-store.ts @@ -0,0 +1,265 @@ +import { ApiError, runs, tasks } from "@trigger.dev/sdk/v3"; +import { del, get, put } from "@vercel/blob"; + +import { readResultPayload } from "@/lib/ai/agent-schema"; +import { describeRunFailure } from "@/lib/ai/run-failure"; +import { prisma } from "@/lib/prisma"; +import { + RECORDED_DECISION_LIMIT, + SPEC_EMPTY_CANVAS_MESSAGE, + SPEC_TOO_LARGE_MESSAGE, + specAgentResultSchema, + specStatsSchema, + type RecordedDecision, + type SpecAgentPayload, + type SpecStatus, +} from "@/lib/spec/spec-schema"; +import type { specAgentTask } from "@/src/trigger/spec-agent"; + +// --------------------------------------------------------------------------- +// Spec storage +// +// One spec per project: the latest Markdown lives in private Vercel Blob +// storage, and the project row holds its URL, the run that produced it, when, +// and its counts. Like AI turns, a finished spec run is stored when the status +// is read, so the task never touches the database or blob storage and a spec +// is saved even if the tab closed mid-run. +// --------------------------------------------------------------------------- + +export type RetrievedSpecRun = + | { state: "running" } + | { state: "missing" } + | { state: "failed"; message?: string } + | { state: "succeeded"; output: unknown; finishedAt: Date }; + +/** External calls, injectable so the storing logic can be tested without Trigger.dev or Blob. */ +export interface SpecStoreDeps { + retrieveRun: (runId: string) => Promise; + uploadMarkdown: (projectId: string, markdown: string) => Promise; + deleteBlob: (url: string) => Promise; +} + +const CANCELLED_MESSAGE = "The spec run was cancelled."; + +async function retrieveFromTrigger(runId: string): Promise { + try { + const run = await runs.retrieve(runId); + if (run.isSuccess) { + return { state: "succeeded", output: run.output, finishedAt: run.finishedAt ?? new Date() }; + } + if (run.isCancelled) { + return { state: "failed", message: CANCELLED_MESSAGE }; + } + if (run.isFailed) { + return { state: "failed", message: run.error?.message }; + } + return { state: "running" }; + } catch (error) { + if (error instanceof ApiError && error.status === 404) { + return { state: "missing" }; + } + console.error("[spec] failed to retrieve run", runId, error); + return { state: "running" }; + } +} + +const defaultDeps: SpecStoreDeps = { + retrieveRun: retrieveFromTrigger, + uploadMarkdown: async (projectId, markdown) => { + const uploaded = await put(`spec-${projectId}.md`, markdown, { + access: "private", + addRandomSuffix: true, + contentType: "text/markdown; charset=utf-8", + }); + return uploaded.url; + }, + deleteBlob: async (url) => { + await del(url); + }, +}; + +const specSelect = { + specMdPath: true, + specRunId: true, + specGeneratedAt: true, + specStats: true, +} as const; + +type ProjectSpecRow = { + specMdPath: string | null; + specRunId: string | null; + specGeneratedAt: Date | null; + specStats: unknown; +}; + +function toStatus( + project: ProjectSpecRow | null, + { pendingRunId = null, lastFailure = null }: { pendingRunId?: string | null; lastFailure?: string | null } = {}, +): SpecStatus { + const stats = specStatsSchema.safeParse(project?.specStats); + const spec = + project?.specMdPath && project.specRunId && project.specGeneratedAt + ? { + runId: project.specRunId, + generatedAt: project.specGeneratedAt.toISOString(), + stats: stats.success ? stats.data : null, + } + : null; + return { spec, pendingRunId, lastFailure }; +} + +/** + * The project's spec status. When the latest spec run finished after the + * stored spec, its Markdown is uploaded and recorded first. The update is + * guarded on the run id, so concurrent reads store a run only once. + */ +export async function getSpecStatus(projectId: string, deps: SpecStoreDeps = defaultDeps): Promise { + const project = await prisma.project.findUnique({ where: { id: projectId }, select: specSelect }); + if (!project) { + return toStatus(null); + } + + const latestRun = await prisma.taskRun.findFirst({ + where: { projectId, kind: "SPEC" }, + orderBy: { createdAt: "desc" }, + select: { runId: true }, + }); + if (!latestRun || latestRun.runId === project.specRunId) { + return toStatus(project); + } + + const run = await deps.retrieveRun(latestRun.runId); + if (run.state === "running") { + return toStatus(project, { pendingRunId: latestRun.runId }); + } + if (run.state === "missing") { + return toStatus(project, { lastFailure: "The spec run could not be found. Try again." }); + } + if (run.state === "failed") { + console.error("[spec] spec-agent run failed", latestRun.runId, run.message); + return toStatus(project, { + lastFailure: describeRunFailure(run.message, { + generic: "Something went wrong generating the spec. Try again.", + timeout: "Spec generation took too long. Try again.", + passthrough: [SPEC_EMPTY_CANVAS_MESSAGE, SPEC_TOO_LARGE_MESSAGE, CANCELLED_MESSAGE], + }), + }); + } + + const parsed = specAgentResultSchema.safeParse(run.output); + if (!parsed.success) { + console.error("[spec] unexpected spec-agent output", latestRun.runId, parsed.error.issues); + return toStatus(project, { lastFailure: "Spec generation returned an unexpected result. Try again." }); + } + + const url = await deps.uploadMarkdown(projectId, parsed.data.markdown); + const { count } = await prisma.project.updateMany({ + where: { id: projectId, OR: [{ specRunId: null }, { specRunId: { not: latestRun.runId } }] }, + data: { + specMdPath: url, + specRunId: latestRun.runId, + specGeneratedAt: run.finishedAt, + specStats: parsed.data.stats, + }, + }); + + if (count === 0) { + // A concurrent read stored this run first; drop the duplicate upload. + await deps.deleteBlob(url).catch(() => undefined); + } else if (project.specMdPath && project.specMdPath !== url) { + await deps.deleteBlob(project.specMdPath).catch(() => undefined); + } + + return toStatus(await prisma.project.findUnique({ where: { id: projectId }, select: specSelect })); +} + +/** Decisions from the user's unexpired AI design sessions in this project, newest first, one per title. */ +export async function loadRecordedDecisions(projectId: string, userId: string): Promise { + const messages = await prisma.aiMessage.findMany({ + where: { + kind: "RESULT", + status: "COMPLETE", + session: { projectId, userId, expiresAt: { gt: new Date() } }, + }, + orderBy: { createdAt: "desc" }, + take: 20, + select: { payload: true }, + }); + + const seen = new Set(); + const decisions: RecordedDecision[] = []; + for (const message of messages) { + for (const decision of readResultPayload(message.payload)?.decisions ?? []) { + const key = decision.title.trim().toLowerCase(); + if (seen.has(key)) { + continue; + } + seen.add(key); + decisions.push({ + title: decision.title, + choice: decision.choice, + rationale: decision.rationale, + alternatives: decision.alternatives, + }); + } + } + return decisions.slice(0, RECORDED_DECISION_LIMIT); +} + +export type StartSpecResult = + | { ok: true; runId: string } + | { ok: false; status: 409; runId: string; error: string } + | { ok: false; status: 502; error: string }; + +/** Starts a spec run for the project, unless one is already in progress. */ +export async function startSpecRun({ + projectId, + userId, + projectName, +}: { + projectId: string; + userId: string; + projectName: string; +}): Promise { + const current = await getSpecStatus(projectId); + if (current.pendingRunId) { + return { ok: false, status: 409, runId: current.pendingRunId, error: "A spec is already being generated" }; + } + + const payload: SpecAgentPayload = { + roomId: projectId, + projectName, + recordedDecisions: await loadRecordedDecisions(projectId, userId), + }; + + let runId: string; + try { + const handle = await tasks.trigger("spec-agent", payload); + runId = handle.id; + } catch (error) { + console.error("[spec] failed to trigger spec-agent", error); + return { ok: false, status: 502, error: "Spec service unavailable" }; + } + + await prisma.taskRun.create({ data: { runId, projectId, userId, kind: "SPEC" } }); + return { ok: true, runId }; +} + +/** The stored spec's Markdown, or null when there is none or it cannot be read. */ +export async function readSpecMarkdown(projectId: string): Promise { + const project = await prisma.project.findUnique({ where: { id: projectId }, select: { specMdPath: true } }); + if (!project?.specMdPath) { + return null; + } + + try { + const result = await get(project.specMdPath, { access: "private", useCache: false }); + if (!result || result.statusCode !== 200) { + return null; + } + return await new Response(result.stream).text(); + } catch (error) { + console.error("[spec] failed to read spec blob", projectId, error); + return null; + } +} diff --git a/prisma/migrations/20260915195650_add_project_spec/migration.sql b/prisma/migrations/20260915195650_add_project_spec/migration.sql new file mode 100644 index 0000000..f172d40 --- /dev/null +++ b/prisma/migrations/20260915195650_add_project_spec/migration.sql @@ -0,0 +1,14 @@ +-- CreateEnum +CREATE TYPE "TaskRunKind" AS ENUM ('DESIGN', 'SPEC'); + +-- AlterTable +ALTER TABLE "Project" ADD COLUMN "specGeneratedAt" TIMESTAMP(3), +ADD COLUMN "specMdPath" TEXT, +ADD COLUMN "specRunId" TEXT, +ADD COLUMN "specStats" JSONB; + +-- AlterTable +ALTER TABLE "TaskRun" ADD COLUMN "kind" "TaskRunKind" NOT NULL DEFAULT 'DESIGN'; + +-- CreateIndex +CREATE INDEX "TaskRun_projectId_kind_createdAt_idx" ON "TaskRun"("projectId", "kind", "createdAt"); diff --git a/prisma/models/project.prisma b/prisma/models/project.prisma index 0ceb925..c07e9f6 100644 --- a/prisma/models/project.prisma +++ b/prisma/models/project.prisma @@ -10,6 +10,11 @@ model Project { description String? status ProjectStatus @default(DRAFT) canvasJsonPath String? + // Latest generated Markdown spec: blob URL, the run that produced it, when, and its counts. + specMdPath String? + specRunId String? + specGeneratedAt DateTime? + specStats Json? createdAt DateTime @default(now()) updatedAt DateTime @updatedAt collaborators ProjectCollaborator[] diff --git a/prisma/models/task-run.prisma b/prisma/models/task-run.prisma index d513e45..95eb52f 100644 --- a/prisma/models/task-run.prisma +++ b/prisma/models/task-run.prisma @@ -1,10 +1,19 @@ +// What a Trigger.dev run was started for. The token route authorises any kind; +// the spec routes look up the latest SPEC run for a project. +enum TaskRunKind { + DESIGN + SPEC +} + model TaskRun { - id String @id @default(cuid()) - runId String @unique + id String @id @default(cuid()) + runId String @unique projectId String userId String - createdAt DateTime @default(now()) + kind TaskRunKind @default(DESIGN) + createdAt DateTime @default(now()) @@index([runId]) @@index([userId, projectId]) + @@index([projectId, kind, createdAt]) } diff --git a/src/trigger/design-agent.ts b/src/trigger/design-agent.ts index 04354a2..646feba 100644 --- a/src/trigger/design-agent.ts +++ b/src/trigger/design-agent.ts @@ -12,6 +12,7 @@ import { resolveDesignModel, } from "@/lib/ai/design-agent-engine"; import { buildCanvasGraph, DESIGN_AGENT_STAGE_KEY, type DesignAgentStage } from "@/lib/design-generation"; +import { readRoomSnapshot } from "@/lib/canvas-room"; import { getLiveblocksClient } from "@/lib/liveblocks"; import { SHAPE_DEFAULTS, type CanvasEdge, type CanvasNode, type CanvasShape } from "@/types/canvas"; @@ -56,13 +57,8 @@ function resolveOrigin(existingNodes: readonly CanvasNode[]): { x: number; y: nu /** Reads the room without changing it, so the agent knows what is already drawn. */ async function readCanvas(roomId: string): Promise { - let snapshot: { nodes: CanvasNode[]; edges: CanvasEdge[] } = { nodes: [], edges: [] }; - - await mutateFlow({ client: getLiveblocksClient(), roomId }, (flow) => { - snapshot = { nodes: [...flow.nodes], edges: [...flow.edges] }; - }); - - return summarizeCanvas(snapshot.nodes, snapshot.edges); + const { nodes, edges } = await readRoomSnapshot(roomId); + return summarizeCanvas(nodes, edges); } /** Generates the diagram for an approved plan and writes it into the room. */ diff --git a/src/trigger/spec-agent.ts b/src/trigger/spec-agent.ts new file mode 100644 index 0000000..bc192db --- /dev/null +++ b/src/trigger/spec-agent.ts @@ -0,0 +1,67 @@ +import { AbortTaskRunError, logger, metadata, task } from "@trigger.dev/sdk/v3"; + +import { modelIdOf, resolveDesignModel } from "@/lib/ai/model"; +import { readRoomSnapshot } from "@/lib/canvas-room"; +import { renderSpecMarkdown } from "@/lib/spec/render-markdown"; +import { generateSpecContent } from "@/lib/spec/spec-engine"; +import { buildSpecGraph, SPEC_MAX_COMPONENTS } from "@/lib/spec/spec-graph"; +import { + SPEC_AGENT_STAGE_KEY, + SPEC_EMPTY_CANVAS_MESSAGE, + SPEC_TOO_LARGE_MESSAGE, + type SpecAgentPayload, + type SpecAgentResult, + type SpecAgentStage, +} from "@/lib/spec/spec-schema"; + +export type { SpecAgentPayload, SpecAgentResult }; + +function setStage(stage: SpecAgentStage) { + metadata.set(SPEC_AGENT_STAGE_KEY, stage); +} + +/** + * Turns the project's canvas into a Markdown technical spec. Reads the room, + * asks the model once for the prose, and renders the document in code. The + * server stores the returned Markdown in blob storage when the status is read. + */ +export const specAgentTask = task({ + id: "spec-agent", + // The model call already retries transient errors; a failed spec is shown to the user instead. + retry: { maxAttempts: 1 }, + maxDuration: 300, + run: async (payload: SpecAgentPayload): Promise => { + setStage("reading"); + const { nodes, edges } = await readRoomSnapshot(payload.roomId); + const graph = buildSpecGraph(nodes, edges); + logger.log("Canvas read for spec", { + roomId: payload.roomId, + components: graph.totalComponents, + connections: graph.connections.length, + groups: graph.groups.length, + recordedDecisions: payload.recordedDecisions.length, + }); + + if (graph.components.length === 0) { + throw new AbortTaskRunError(SPEC_EMPTY_CANVAS_MESSAGE); + } + if (graph.totalComponents > SPEC_MAX_COMPONENTS) { + throw new AbortTaskRunError(SPEC_TOO_LARGE_MESSAGE); + } + + setStage("writing"); + const model = resolveDesignModel(); + const content = await generateSpecContent(model, graph, payload.projectName, payload.recordedDecisions); + + setStage("rendering"); + const { markdown, stats } = renderSpecMarkdown(graph, content, { + projectName: payload.projectName, + generatedAt: new Date(), + modelId: modelIdOf(model), + }); + logger.log("Spec rendered", { ...stats, characters: markdown.length }); + + setStage("done"); + return { markdown, stats }; + }, +});