From e0b67bb804a3c180928771a3a8eb6da95f414fdd Mon Sep 17 00:00:00 2001 From: Deon Menezes Date: Sat, 29 Aug 2026 17:34:14 -0700 Subject: [PATCH 1/5] Document what EditAI does, and give it a page that shows it The README described the timeline tools but not the thing people ask about first: what you can say to it, and what happens at each of the three layers the agent drives (your UI, ours, the ffmpeg encode). It now carries the full 16-tool reference with arguments, the connector story including how to attach an MCP server mid-conversation, both sandboxes and what each is for, the environment table, test counts, and a "Known limits" section, because export writes a render description rather than encoding video and the demo does not make that obvious. The landing page in site/ renders the agent server's own sample project and performs a real ripple delete on it: mapTime() applies the same rule project.ts does, so clips straddling a silence get shorter rather than merely shifting, and 24.0s becomes 21.1s. Its palette is lifted from the editor's stylesheet and favicon rather than invented, so the page and the product read as one thing. Two numbers in the README were wrong and are corrected here: the silence ranges in the walkthrough were sketched rather than read off project.ts (they are 3.2-4.1, 9.6-10.4, 16.8-18.0), and .env.example lives in apps/agent, not the repo root. --- README.md | 398 ++++++++++++++------ site/.gitignore | 1 + site/README.md | 37 ++ site/index.html | 918 +++++++++++++++++++++++++++++++++++++++++++++++ site/vercel.json | 13 + 5 files changed, 1264 insertions(+), 103 deletions(-) create mode 100644 site/.gitignore create mode 100644 site/README.md create mode 100644 site/index.html create mode 100644 site/vercel.json diff --git a/README.md b/README.md index f8d0f28..47f8cc8 100644 --- a/README.md +++ b/README.md @@ -1,10 +1,28 @@ +
+ # EditAI -**An AI video editor you talk to.** Say "remove the silences" or "caption every clip", and an agent -makes the edit on your real timeline: it reads the project, decides which cuts to make, and applies -them. Anything destructive stops and asks you first. +**An AI harness for video editing. You describe the edit; an agent makes it on your real timeline.** + +[**Live site**](https://editai-agent.vercel.app) · +[Quick start](#getting-started) · +[The 16 tools](#the-16-timeline-tools) · +[Add your own MCP](#reaching-past-the-timeline) · +[Review evidence](#code-review-evidence-qodo) [![License: MIT](https://img.shields.io/badge/license-MIT-green?style=flat)](LICENSE) +[![Built on TrueForge](https://img.shields.io/badge/harness-TrueForge-7c5cff?style=flat)](https://trueforge.dev) +[![MCP](https://img.shields.io/badge/protocol-MCP-7c5cff?style=flat)](https://modelcontextprotocol.io) +[![Tests](https://img.shields.io/badge/tests-31%20passing-3aa39b?style=flat)](#tests) +[![Reviewed by Qodo](https://img.shields.io/badge/reviewed%20by-Qodo-e0a63b?style=flat)](#code-review-evidence-qodo) + +
+ +--- + +Say "remove the silences" or "caption every clip", and an agent makes the edit on your real +timeline: it reads the project, decides which cuts to make, and applies them. Anything destructive +stops and asks you first. The tedious parts of editing are the ones a machine should do. Cutting dead air out of a twenty-minute take is thirty minutes of scrubbing; here it is one sentence and one approval click. @@ -12,39 +30,57 @@ Captioning every clip means transcribing each one by hand; here a sub-agent hand parallel and the captions land on their own track, timed. The agent does the mechanical work, and you keep the decisions: it proposes, you approve, and every edit is undoable. -### What it can do today +**Three layers, one conversation.** Depending on what you ask for, the same agent drives your own +editing UI, the editor in this repo, or the ffmpeg encode underneath it. You do not choose the +layer; the request does. -| Ask for this | What the agent does | -| --- | --- | -| *"Remove the silences"* | Finds every silent range on the voice track and ripple-deletes it across all tracks, closing the gaps. Verified: 24s → 21.1s, exactly the 2.9s of silence. | -| *"Caption every video clip"* | Fans out one sub-agent per clip to transcribe in parallel, merges the results, and lays timed captions on a new track. | -| *"Cut the intro to 3 seconds"* | Trims the clip, keeping the media in sync by moving its source offset. | -| *"Duck the music under the voiceover"* | Sets clip volume. | -| *"Split this at 15 seconds"* | Cuts a clip in two, both halves still frame-accurate. | -| *"Export it at 1080p"* | Renders, after you approve. | +--- -### The agent tools +## Watch it work -The timeline is exposed to the agent as **16 [MCP](https://modelcontextprotocol.io) tools**, not as -a prompt describing a timeline. The agent calls real functions against real state: +The one operation worth showing end to end, because the numbers are checkable: + +``` +you ▸ remove the silences -- **Read:** `get_project`, `transcribe_clip`, `find_silences`, `detect_beats`, `list_changes` -- **Write:** `split_clip`, `trim_clip`, `move_clip`, `set_volume`, `add_text`, `add_captions`, `undo` -- **Destructive:** `delete_clip`, `ripple_delete`, `remove_silences` -- **Gated:** `export_project` +agent → find_silences(min_duration: 0.5, track_id: "a1") + ← 3 ranges: 3.2-4.1, 9.6-10.4, 16.8-18.0 (2.9s total) -Every tool validates its input and returns errors to the model as data, so a stale clip id becomes -a correction the agent recovers from rather than a failed turn. The destructive four are published -with MCP's `destructiveHint` annotation, which is what makes the harness stop and ask you before -they run. + "Three silent ranges, 2.9s in all. This will cut every track. Remove them?" -Because it is MCP, the same agent can reach anything else that speaks MCP: web search, your issue -tracker, an internal API you wrap yourself. Connectors attach by name and authorize in chat. +you ▸ [Approve] + +agent → remove_silences(min_duration: 0.5) + ← duration 24.0s → 21.1s, every track cut, gaps closed + + "Cut 2.9s of dead air across every track. Say undo to put it back." +``` + +24.0 down to 21.1 is exactly the 2.9 seconds of silence, with picture, sound and captions still in +sync. There is an animated version of this on the [live site](https://editai-agent.vercel.app). + +## What you can ask for + +| Ask | What the agent actually does | Tools | +| --- | --- | --- | +| *"Remove the silences"* | Finds every silent range on the voice track, tells you what will go, then ripple-deletes it across all tracks, closing the gaps. Verified: 24s → 21.1s. | `find_silences` → `remove_silences` | +| *"Caption every video clip"* | Fans out one sub-agent per clip to transcribe in parallel, merges the results (including sentences that straddle a cut), and lays timed captions on their own track. | `transcribe_clip` ×N → `add_captions` | +| *"Cut this to the beat"* | Reads the tempo off the music track and splits on the beat grid. Both halves stay frame-accurate. | `detect_beats` → `split_clip` | +| *"Trim the intro to 3 seconds"* | Trims the clip and moves its source offset by the same amount, so the picture does not jump. | `trim_clip` | +| *"Duck the music under the voiceover"* | Sets clip volume where the voice track is speaking and restores it where it is not. | `get_project` → `set_volume` | +| *"Grade it warmer and add grain"* | Writes the ffmpeg filter graph, runs it in the media sandbox, then probes the output to confirm it matches the intent. | `probe_media` → `run_ffmpeg` | +| *"Kill the room tone"* | Denoises the voice track in the sandbox, leaving the original file untouched beside it. | `run_ffmpeg` (`afftdn`) | +| *"Put their logo in the corner"* | Searches the live web through Bright Data, scrapes the asset, and brings it into the project. | `search_engine` → `scrape_as_markdown` | +| *"Export it at 1080p"* | Renders, after you approve. | `export_project` | + +Motion graphics, transitions and animation work the same way: either as an ffmpeg filter graph in +the media sandbox, or by attaching an MCP server that specialises in them. See +[Reaching past the timeline](#reaching-past-the-timeline). ## How it works -EditAI is built on [TrueForge](https://trueforge.dev), an open-source agent harness. The harness -runs the agent loop; EditAI supplies the domain. +EditAI is built on [TrueForge](https://trueforge.dev), TrueFoundry's open-source agent harness. The +harness runs the agent loop; EditAI supplies the domain. ``` browser harness domain @@ -53,7 +89,15 @@ runs the agent loop; EditAI supplies the domain. │ (apps/web) │ + SSE │ agent loop │ │ 16 timeline │ │ │ │ approvals │ │ tools │ │ timeline ◄──┼───────────┼──────────────┼── SSE ───┤ project store │ -└──────────────┘ └──────────────┘ └──────────────────┘ +└──────────────┘ └──────┬───────┘ └──────────────────┘ + │ MCP + ┌─────────────┼──────────────┐ + ▼ ▼ ▼ + ┌──────────────┐ ┌─────────┐ ┌──────────────┐ + │ ffmpeg │ │ bright- │ │ anything │ + │ sandbox │ │ data │ │ else that │ + │ (container) │ │ (web) │ │ speaks MCP │ + └──────────────┘ └─────────┘ └──────────────┘ ``` The editor never calls a model. It creates a session, streams turn events, renders tool calls and @@ -68,22 +112,135 @@ server's event stream, so an edit the agent makes shows up in the UI as it happe | Sessions that survive a reload | `apps/web/src/components/editor/use-assistant.ts` replays turns and re-attaches to a running one | | Live web research | `bright-data` connector: `search_engine` and `scrape_as_markdown`, attached deferred so it costs no context until a task needs it (see [docs/brightdata.md](docs/brightdata.md)) | | Any model provider | `apps/agent/scripts/setup.ts` registers whichever API keys are present, including any OpenAI-compatible endpoint | -| Sandboxed execution | Two layers: the harness sandbox (Daytona, configured automatically when `DAYTONA_API_KEY` is set) for general code, and `packages/ffmpeg-sandbox` for media work | +| Sandboxed execution | Two layers: the harness sandbox (Daytona) for general code, and `packages/ffmpeg-sandbox` for media work | +| Domain know-how | `skills/video-editing/SKILL.md`, loaded by the agent whenever a task touches ffmpeg | + +## The 16 timeline tools + +The timeline is exposed to the agent as **16 [MCP](https://modelcontextprotocol.io) tools**, not as +a prompt describing a timeline. The agent calls real functions against real state. + +| Tool | Kind | Arguments | What it does | +| --- | --- | --- | --- | +| `get_project` | read | | Tracks, clips, media metadata, exports. Call it first; ids change after edits. | +| `list_changes` | read | `limit` | Recent edits, oldest first. | +| `transcribe_clip` | read | `clip_id` | Speech inside one clip, as timed segments in timeline seconds. | +| `find_silences` | read | `min_duration`, `track_id` | Silent ranges on a track. Preview only. | +| `detect_beats` | read | `track_id` | Beat timestamps derived from the track's tempo. | +| `split_clip` | write | `clip_id`, `at` | Cuts a clip in two. Returns both halves. | +| `trim_clip` | write | `clip_id`, `start?`, `end?` | New in/out points, keeping media in sync via `sourceOffset`. | +| `move_clip` | write | `clip_id`, `start?`, `track_id?` | New start time, or another track of the same kind. | +| `set_volume` | write | `clip_id`, `volume` | Clip volume, 0 to 100. | +| `add_text` | write | `text`, `start`, `duration`, `track_id` | A title or caption on a text track. | +| `add_captions` | write | `segments[]`, `track_label` | Timed captions on the captions track, creating it if needed. | +| `undo` | write | | Reverts the most recent change. | +| `delete_clip` | **destructive** | `clip_id` | Removes a clip, leaving a gap. | +| `ripple_delete` | **destructive** | `start`, `end` | Removes a range from every track and closes the gap. | +| `remove_silences` | **destructive** | `min_duration`, `track_id` | Ripple-deletes every silence over the threshold. | +| `export_project` | **approval** | `format`, `resolution` | Renders the timeline to a file. | + +Every tool validates its input with zod and returns errors to the model **as data**, so a stale clip +id becomes a correction the agent recovers from rather than a failed turn. + +The four gated tools are published with MCP's `destructiveHint` annotation, and the agent declares +`require_approval_for_tools: ["@destructive", "export_project"]`. That turns the annotation into a +pause: the harness stops the turn, the editor shows the tool and its arguments, and the run only +continues once a person allows or denies it. + +Three semantics worth knowing before you read the code: + +- **`sourceOffset` keeps media in sync.** Trimming a clip's start moves its offset into the source + file by the same amount, so the picture does not jump. +- **Ripple delete is the interesting operation.** Removing a range cuts every track, splits any clip + straddling the range, and shifts everything after it left. `remove_silences` applies it once per + silence, from the end backwards, so earlier ranges stay valid. +- **Captions merge across clip boundaries.** Fanning captioning out per clip means a sentence + spanning a cut is reported twice, clamped to each side; `add_captions` merges those back into one. + +See [apps/agent/README.md](apps/agent/README.md) for the full tool reference and timeline semantics. + +## Reaching past the timeline + +Because everything is MCP, the same agent can reach anything else that speaks MCP: web search, a +motion-graphics server, your issue tracker, an internal API you wrap yourself. **Adding a capability +is a name in a list and a restart, not a release.** + +```bash +# keyless, the default +EDITAI_CONNECTORS=exa bun run setup + +# header auth +BRIGHT_DATA_MCP_HEADER="Authorization: Bearer " \ + EDITAI_CONNECTORS=exa,bright-data bun run setup + +# OAuth: dynamic client registration, nothing to configure here. +# The first time the agent reaches for it, the turn pauses with an authorize +# URL and the editor shows a Connect button. Verified against Linear. +EDITAI_CONNECTORS=exa,linear bun run setup + +# your own server, attached the same way +EDITAI_CONNECTORS=exa,bright-data,motion-graphics bun run setup +``` + +Extra connectors attach **read-only and deferred**, so a connector you rarely use costs nothing in +context until the agent actually reaches for it. The tools go live on the agent's next turn, in the +same conversation. + +**Bright Data** is the one wired up and verified: `search_engine`, `search_engine_batch`, +`scrape_as_markdown`, `scrape_batch` and `ask_brightdata_assistant`. It is what lets you say "put +their logo in the corner" and get the current logo rather than a model's memory of one. +Setup details and the auth gotcha are in [docs/brightdata.md](docs/brightdata.md). + +## Where the code runs + +Generated code never runs on the host. Two layers cover the two kinds of work. + +**Harness sandbox (Daytona).** `setup.ts` registers the provider when `DAYTONA_API_KEY` is present, +and the agent's `exec` tool then runs in a remote sandbox. Verified end to end against the running +harness: + +``` +sandbox.created sandbox_id: v1:daytona:default.e17058ee-... +exec python3 -c "print(sum(int(x)**2 for x in range(1,101)))" +tool.response {"success":true,"response":{"exitCode":0,"result":"338350\n"}} +``` + +**Media sandbox (`packages/ffmpeg-sandbox`).** ffmpeg and ffprobe are not safe to point at +agent-supplied arguments on the host, so every invocation runs in a throwaway container: +`--network none`, capped memory and CPU, `--pids-limit 256`, a non-root user, a single mounted +workspace, and a hard timeout. It exposes four tools: + +| Tool | What it does | +| --- | --- | +| `list_media` | What is in the workspace | +| `probe_media` | Duration, resolution, codecs, fps | +| `run_ffmpeg` | Runs ffmpeg with an argument array | +| `run_python` | Glue work, parsing, arithmetic | + +`run_python` is annotated `destructiveHint: true`, so the harness shows the script and waits for +approval before it runs: a script with the workspace mounted read-write can delete the source media, +and the container is not a defence against that. + +The split is deliberate. The harness sandbox is for computation the agent should not estimate; the +media sandbox is for work that must reach the media files. ## Getting started -You need [Bun](https://bun.sh) and Node 22.14+ (for the harness). +You need [Bun](https://bun.sh) and Node 22.14+ (for the harness). Docker is needed only for the +media sandbox. ```bash +git clone https://github.com/deonmenezes/edit-ai.git +cd edit-ai bun install -# 1. the harness +# 1. the harness (separate terminal) npx @truefoundry/trueforge@latest # http://localhost:8790 # 2. the timeline tools cd apps/agent && bun run start # http://localhost:8941 -# 3. wire them together (any one key is enough) +# 3. wire them together (any one model key is enough) ANTHROPIC_API_KEY=sk-... bun run setup # 4. the editor @@ -93,24 +250,48 @@ cd ../.. && bun run dev:web # http://localhost:5173 Then ask for an edit: "Remove the silences", "Caption every video clip". Without the harness running, the editor still loads with a sample timeline; the assistant panel -says it is offline. +says it is offline. `setup` is idempotent, so rerun it after changing `agent.json` or adding a key. -## Stack +### Environment -- [TanStack Start](https://tanstack.com/start) + React 19, Vite, Tailwind CSS v4, shadcn/ui -- [TrueForge](https://trueforge.dev) agent harness, [MCP](https://modelcontextprotocol.io) tools -- Cloudflare Workers (via Wrangler), Bun + Turborepo monorepo +| Variable | Effect | +| --- | --- | +| `ANTHROPIC_API_KEY` / `OPENAI_API_KEY` / `GEMINI_API_KEY` | Registers that provider with every model in the TrueForge catalog. | +| `NVIDIA_API_KEY` | Registers NVIDIA NIM as a custom OpenAI-compatible provider. | +| `OPENAI_COMPATIBLE_BASE_URL` + `_API_KEY` + `_MODELS` (+ `_NAME`) | Registers any other OpenAI-compatible endpoint: vLLM, Ollama, a gateway. | +| `DAYTONA_API_KEY` | Configures the harness sandbox and enables it on the agent. | +| `EDITAI_MODEL` | Pins the agent's model instead of picking the best configured one. | +| `EDITAI_CONNECTORS` | Extra MCP servers to attach, comma separated. Defaults to `exa`. | +| `_MCP_HEADER` | Credential for a header-auth connector, e.g. `BRIGHT_DATA_MCP_HEADER="Authorization: Bearer ..."`. | +| `TRUEFORGE_BASE_URL` | Defaults to `http://localhost:8790`. | +| `EDITAI_AGENT_PORT` | Defaults to `8941`. | + +At least one model key is required. With none set, `setup` stops and tells you. `setup` refuses to +POST keys over plaintext HTTP to anything but localhost. + +Copy [`apps/agent/.env.example`](apps/agent/.env.example) to `apps/agent/.env` to start from a +documented set. ## Layout ``` apps/ - web/ the editor: timeline, preview, assistant panel - agent/ MCP server exposing the timeline, plus the agent definition + web/ the editor: timeline, preview, assistant panel + agent/ MCP server exposing the timeline, plus the agent definition + src/tools.ts the 16 tools + src/project.ts the timeline model: split, trim, ripple delete, captions + scripts/setup.ts registers models, connectors, sandbox and the agent +packages/ + ffmpeg-sandbox/ containerised ffmpeg/ffprobe/python MCP server +skills/ + video-editing/ ffmpeg recipes and rules the agent loads on demand +docs/ + brightdata.md connector setup and verification +site/ the landing page (static, deployed to Vercel) +tests/ the Qodo verdict parser's fixtures and tests +.github/workflows/ CI, and the Qodo merge gate ``` -See [apps/agent/README.md](apps/agent/README.md) for the tool reference and timeline semantics. - ## Scripts | Command | What it does | @@ -118,73 +299,65 @@ See [apps/agent/README.md](apps/agent/README.md) for the tool reference and time | `bun run dev` | Run every app in dev mode | | `bun run dev:web` | Run only the web app | | `bun run build` | Build every app | +| `bun run test` | Run the Bun test suites | | `bun run deploy` | Build and deploy to Cloudflare | -## Sandboxed execution +## Tests -Generated code never runs on the host. Two layers cover the two kinds of work. +**31 tests**, all passing. -**Harness sandbox (Daytona).** `setup.ts` registers the provider when -`DAYTONA_API_KEY` is present, and the agent's `exec` tool then runs in a remote -sandbox. Verified end to end against the running harness: +| Suite | Count | Covers | +| --- | --- | --- | +| `apps/agent/test/project.test.ts` | 17 | Split and trim invariants, ripple-delete arithmetic across tracks, silence removal, transcript windowing, caption merging, undo, and the clip-id and trim-bound regressions Qodo surfaced. | +| `apps/web/.../transcript.test.ts` | 6 | Transcript windowing and caption rendering in the editor. | +| `tests/test_qodo_verdict.py` | 8 | The merge gate's verdict parser, against the real Qodo comment bodies that broke it. | +```bash +bun run test # the Bun suites (agent + web) +cd apps/agent && bun test # the timeline model +python3 -m pytest tests/ # the Qodo verdict parser ``` -sandbox.created sandbox_id: v1:daytona:default.e17058ee-... -exec python3 -c "print(sum(int(x)**2 for x in range(1,101)))" -tool.response {"success":true,"response":{"exitCode":0,"result":"338350\n"}} -``` -**Media sandbox (`packages/ffmpeg-sandbox`).** ffmpeg and ffprobe are not safe to -point at agent-supplied arguments on the host, so every invocation runs in a -throwaway container: `--network none`, capped memory and CPU, `--pids-limit 256`, -a non-root user, a single mounted workspace, and a hard timeout. `run_python` -there is annotated `destructiveHint: true`, so the harness shows the script and -waits for approval before it runs, because a script with the workspace mounted -read-write can delete the source media and the container is not a defence -against that. - -The split is deliberate: the harness sandbox is for computation the agent should -not estimate, and the media sandbox is for work that must reach the media files. - -## Qodo Code Review Evidence - -### Merging on the review - -Qodo posts its verdict as an issue comment. It publishes no check run, no commit -status and no approving review, so GitHub's own auto-merge has nothing to gate -on. `.github/workflows/qodo-automerge.yml` is that missing gate. - -It listens for edited comments as well as created ones, because Qodo posts a -placeholder and then edits the verdict into that same comment: a gate watching -only for new comments never sees a verdict at all. On an edited event -`comment.user` is still the bot even when a person did the editing, so the -sender is checked too. - -It fails closed in every direction, because the first draft did not and Qodo -said so. It reads only the structured counter chips, never the prose: Qodo -quotes findings and diff hunks verbatim, so the words "no issues found" appear -inside reviews that are *not* clean, and any substring test on the comment body -is forgeable by the pull request's own content. It binds the verdict to the -commit Qodo footers in the comment and refuses to merge when that is no longer -the head, since a review applies to one revision and `issue_comment` runs give a -job no link to the pull request head. It treats a failure to read check state as -an error rather than as an absence of failures. And it merges with -`--match-head-commit`, so a push racing the merge is rejected by GitHub instead -of slipping in. +## Code review evidence (Qodo) + +Every pull request here is reviewed by [Qodo](https://qodo.ai). **Eight findings across three +reviews: seven were real and are fixed, one was checked and rejected with a proof.** + +### The merge gate + +Qodo posts its verdict as an issue comment. It publishes no check run, no commit status and no +approving review, so GitHub's own auto-merge has nothing to gate on. +[`.github/workflows/qodo-automerge.yml`](.github/workflows/qodo-automerge.yml) is that missing gate. + +It listens for edited comments as well as created ones, because Qodo posts a placeholder and then +edits the verdict into that same comment: a gate watching only for new comments never sees a verdict +at all. On an edited event `comment.user` is still the bot even when a person did the editing, so +the sender is checked too. + +It fails closed in every direction, because the first draft did not and Qodo said so: + +- It reads only the **structured counter chips**, never the prose. Qodo quotes findings and diff + hunks verbatim, so the words "no issues found" appear inside reviews that are *not* clean, and any + substring test on the comment body is forgeable by the pull request's own content. +- It **binds the verdict to the commit** Qodo footers in the comment and refuses to merge when that + is no longer the head, since a review applies to one revision and `issue_comment` runs give a job + no link to the pull request head. +- It treats a **failure to read check state as an error**, not as an absence of failures. +- It merges with **`--match-head-commit`**, so a push racing the merge is rejected by GitHub instead + of slipping in. ### PR #3, first review -[PR #3](https://github.com/deonmenezes/edit-ai/pull/3) was reviewed with Qodo Merge, which raised -three issues. All three were real and all three are fixed in the PR: +[PR #3](https://github.com/deonmenezes/edit-ai/pull/3) raised three issues. All three were real and +all three are fixed: 1. **Duplicate clip ids after repeated ripple deletes** (`apps/agent/src/project.ts`). The - right-hand half of a split clip took the id `${c.id}r`, so a clip cut more than once produced - the same id twice. Reproduced on the sample project: `removeSilences` left three clips sharing - `c6r` and three sharing `c7r`, which breaks clip lookup, deletion and React keys. Ids are now - allocated from the set of ids in use. Covered by three regression tests. + right-hand half of a split clip took the id `${c.id}r`, so a clip cut more than once produced the + same id twice. Reproduced on the sample project: `removeSilences` left three clips sharing `c6r` + and three sharing `c7r`, which breaks clip lookup, deletion and React keys. Ids are now allocated + from the set of ids in use. Covered by three regression tests. 2. **Export announced before the file existed.** `exportProject` committed the record, which - notifies SSE subscribers synchronously, and only then wrote the file. The write now happens - first. + notifies SSE subscribers synchronously, and only then wrote the file. The write now happens first. 3. **The session-restore effect leaked its stream on unmount**, calling `setState` on a gone component and leaving the connection open. Its cleanup now aborts the controller. @@ -201,10 +374,10 @@ rejected: - **Real:** `setup.ts` POSTs API keys to the harness, so it now refuses to do that over plaintext HTTP to anything but localhost. - **False positive:** the reviewer called the source-media bound in `trimClip` - (`end - (c.start - c.sourceOffset)`) wrong and predicted a clip could be extended to 28s instead - of 18s. The expression expands to `c.sourceOffset + (end - c.start)`, which is the correct source - time, and the code accepts exactly up to the limit and rejects one frame past it. Two tests now - pin that boundary so the correct form is not "fixed" into a broken one later. + (`end - (c.start - c.sourceOffset)`) wrong and predicted a clip could be extended to 28s instead of + 18s. The expression expands to `c.sourceOffset + (end - c.start)`, which is the correct source + time, and the code accepts exactly up to the limit and rejects one frame past it. Two tests now pin + that boundary so the correct form is not "fixed" into a broken one later. ### PR #2 review @@ -216,14 +389,33 @@ rejected: workspace is exactly what it is meant to reach. The tool is now registered with `destructiveHint: true`, so the harness stops and shows the script for approval first, the same gate the timeline's `delete_clip` and `ripple_delete` use. -- **The path guard rejected valid code.** Scanning a Python source string for `..` or a URL - rejected correct programs without adding a boundary. Path validation now applies only to - path-like arguments; for code payloads the container is the boundary. +- **The path guard rejected valid code.** Scanning a Python source string for `..` or a URL rejected + correct programs without adding a boundary. Path validation now applies only to path-like + arguments; for code payloads the container is the boundary. + +## Stack + +- [TanStack Start](https://tanstack.com/start) + React 19, Vite, Tailwind CSS v4, shadcn/ui +- [TrueForge](https://trueforge.dev) agent harness, [MCP](https://modelcontextprotocol.io) tools +- Cloudflare Workers (via Wrangler), Bun + Turborepo monorepo +- Landing page: hand-written static HTML in `site/`, deployed to Vercel + +## Known limits + +Worth stating plainly, because the demo does not make them obvious: +- `export_project` writes a JSON description of the render rather than encoding video. Wiring it to + ffmpeg is the obvious next step and does not change the agent-facing contract. +- Transcripts, silences and tempo come from media metadata in the sample project rather than from + running ASR and onset detection over real files. +- Motion graphics have no dedicated timeline tool yet. Today they go through the ffmpeg sandbox or + an attached MCP server. ## Contributing -Issues and pull requests are welcome. See [CONTRIBUTING.md](.github/CONTRIBUTING.md) for setup and guidelines, and open an issue first for anything larger than a bug fix. +Issues and pull requests are welcome. See [CONTRIBUTING.md](.github/CONTRIBUTING.md) for setup and +guidelines, and open an issue first for anything larger than a bug fix. Every PR is reviewed by Qodo +before it can merge. ## License diff --git a/site/.gitignore b/site/.gitignore new file mode 100644 index 0000000..e985853 --- /dev/null +++ b/site/.gitignore @@ -0,0 +1 @@ +.vercel diff --git a/site/README.md b/site/README.md new file mode 100644 index 0000000..e4d7e00 --- /dev/null +++ b/site/README.md @@ -0,0 +1,37 @@ +# site + +The EditAI landing page: + +One hand-written `index.html` with inline CSS and JS. No build step, no dependencies, no +framework. Open the file to work on it. + +```bash +python3 -m http.server 4477 --directory site # http://localhost:4477 +``` + +## The hero timeline + +The timeline in the hero is not a screenshot. It renders the same sample project the agent server +ships with (`apps/agent/src/project.ts`: `intro.mp4`, `b-roll.mp4`, `talking-head.mp4`, the +voiceover and its three silences) and performs a real ripple delete on it: `mapTime()` in the page +script applies the same rule the server does, so 24.0s becomes 21.1s and clips straddling a silence +get shorter rather than merely shifting. + +**If the sample project changes, change `SILENCES` and `LANES` in `index.html` to match.** The page +claims those are real numbers, so they have to stay real. + +## Colours + +Every colour is lifted from the editor rather than invented: the well, panel and foreground greys +from `apps/web/src/styles.css`, the violet `#7c5cff` from the favicon, and the three clip colours +(`#3aa39b` video, `#e0a63b` text, `#5fae63` audio) that the timeline paints tracks with. + +## Deploying + +```bash +cd site +vercel deploy --prod --scope deonmenezes-projects +``` + +Production aliases: `editai-agent.vercel.app` (canonical), `edit-ai-video.vercel.app`, +`edit-ai-lemon.vercel.app`. diff --git a/site/index.html b/site/index.html new file mode 100644 index 0000000..a61f4f6 --- /dev/null +++ b/site/index.html @@ -0,0 +1,918 @@ + + + + + +EditAI: talk to your timeline + + + + + + + + + + + + + + + + +
+ + + EditAI + + 00:00:00:00 + + + GitHub +
+ + + + +
+
+ + +
+
+ 00:00:00 + Cold open + +
+ +

Talkto yourtimeline

+ +

+ EditAI is an AI harness for video editing. Ask for the cut, the motion + graphic, the filter or the noise gate. An agent makes the edit on your real timeline + through 16 MCP tools, and stops for your approval before anything destructive. +

+ + +
+
+ + + ⌘⏎ +
+ +
+
V1
+
A1
+
T1
+
+ +
+ Waiting for an instruction + + + 24.0s +
+
+ +

+ Real behaviour, real numbers: find_silences then + ripple_delete across every track, 24.0s down to 21.1s, which is + exactly the 2.9 seconds of dead air. Nothing is cut until you approve it, and every edit undoes. +

+
+ + +
+
+ 00:00:14 + The handoff + +
+

Connect the MCP.
The agent takes the room.

+

+ Point your client at EditAI's MCP server and an agent comes up inside a sandbox on + TrueForge, TrueFoundry's open-source agent harness. From there it drives the + edit at whichever layer the job needs. +

+ +
+
+ Layer one +

Your editing UI

+

Bring your own front end. The timeline is exposed as tools, not as a prompt describing a + timeline, so any MCP client can drive it.

+
any MCP client
+
+
+ Layer two +

Our editor

+

A full timeline, preview and assistant panel. The agent's edits stream in over SSE, so + the tracks redraw as the work happens.

+
apps/web · TanStack Start
+
+
+ Layer three +

The encode itself

+

When a job needs the real file, the agent writes the ffmpeg filter graph and runs it in a + locked-down container.

+
ffmpeg · ffprobe · python
+
+
+ +
# 1 · the harness
+npx @truefoundry/trueforge@latest          → localhost:8790
+
+# 2 · the timeline tools
+cd apps/agent && bun run start            → localhost:8941
+
+# 3 · wire them together (any one key is enough)
+ANTHROPIC_API_KEY=sk-... bun run setup
+
+# 4 · the editor
+bun run dev:web                           → localhost:5173
+
+ + +
+
+ 00:00:36 + Say it + +
+

Ask in a sentence.
Watch the tracks move.

+

+ Every ask below runs real tools against real state. Clip ids, source offsets and frame + boundaries are the agent's problem, not yours. +

+ +
+
+

▸Remove the silences

+

Finds every silent range on the voice track, tells you what will go, and + ripple-deletes it across all tracks so the gaps close.

+

find_silences → remove_silences

+
+
+

▸Caption every clip

+

Fans out one sub-agent per clip to transcribe in parallel, merges the + results, and lays timed captions on their own track.

+

transcribe_clip × N → add_captions

+
+
+

▸Cut this to the beat

+

Reads the tempo off the music track and splits on the beat grid, both + halves still frame-accurate.

+

detect_beats → split_clip

+
+
+

▸Duck the music under the voiceover

+

Sets clip volume where the voice track is speaking and puts it back + where it is not.

+

get_project → set_volume

+
+
+

▸Grade it warmer and add grain

+

Writes the ffmpeg filter graph, runs it in the media sandbox, then probes + the output to confirm it is what you asked for.

+

probe_media → run_ffmpeg

+
+
+

▸Kill the room tone

+

Denoises the voice track in the sandbox and leaves the original file + untouched next to it.

+

run_ffmpeg · afftdn

+
+
+ +

+ Sixteen timeline tools in all. Five read, seven write and undo, three destructive, one gated + export. The destructive ones carry MCP's destructiveHint, which is + what makes the harness stop and ask you first. +

+
+ + +
+
+ 00:00:58 + Pull it in + +
+

Need a logo?
It goes and gets one.

+

+ The agent reaches the live web through Bright Data, so a logo, a product shot + or a reference frame is one sentence away from being on your timeline. The connector attaches + deferred: it costs no context until a task actually needs it. +

+
+
+ Find it +

search_engine

+

Live search results, not a training-set memory of what a brand looked like two years ago.

+
+
+ Take it +

scrape_as_markdown

+

Pulls the page down clean, so the agent can lift the asset and the copy around it.

+
+
+
+ + +
+
+ 00:01:16 + Add a tool mid-call + +
+

A new skill is a
line in a config.

+ +
+
+ On the call + Bro, can you add an MCP for motion graphics? I want to use it in this video. +
+
+ You + Sent to your agent. +
+
+ +

+ Every capability here is an MCP server, so adding one is a name in a list and a restart, not a + release. The agent picks up the new tools on its next turn and starts using them in the same + conversation. +

+ +
EDITAI_CONNECTORS=exa,bright-data,motion-graphics bun run setup
+
+✓ motion-graphics   attached   read-only, deferred
+✓ agent updated     14 connectors → tools live on the next turn
+ +

+ Catalog servers attach by name, OAuth ones authorise in chat, and anything you host yourself + attaches the same way. Read-only and deferred by default, so a connector you rarely use does + not crowd the agent's context. +

+
+ + +
+
+ 00:01:34 + Where it runs + +
+

Nothing the agent
writes runs on your box.

+

+ Two sandboxes, because the two kinds of work need different boundaries. +

+
+
+ Daytona +

Harness sandbox

+

For the computation the agent should not estimate: arithmetic, parsing, anything it + would otherwise guess at. Runs remotely, registered the moment the key is present.

+
exit 0 · 338350
+
+
+ Container +

Media sandbox

+

Every ffmpeg call runs in a throwaway container: no network, capped memory and CPU, 256 + pids, non-root, one mounted workspace, hard timeout. Python there asks for approval first, + because a script with the workspace mounted can delete your source media.

+
--network none · --pids-limit 256
+
+
+
+ + +
+
+ 00:01:52 + Reviewed by Qodo + +
+

Every line here was
reviewed by Qodo.

+

+ Not a badge. Eight findings across three reviews, each one either fixed or answered with a + proof, and a merge gate that reads Qodo's structured verdict rather than its prose. +

+ +
+ 8Findings raised + 7Real, fixed + 1Checked, rejected + 5Regression tests added +
+ +
+
+ PR #3 + Duplicate clip ids after repeated ripple deletes. A clip + cut twice produced the same id twice, breaking lookup, deletion and React keys. The test + suite had missed it because the existing test asserted the buggy id as correct. + FIXED +
+
+ PR #3 + Export announced before the file existed. The record was + committed, notifying subscribers, and only then was the file written. + FIXED +
+
+ PR #3 + Session restore leaked its stream on unmount. It set state + on a gone component and left the connection open. + FIXED +
+
+ PR #3 + Setup posted API keys over plaintext HTTP. It now refuses + to do that to anything but localhost. + FIXED +
+
+ PR #3 + The trim bound was called wrong, and was not. The source + expression expands to the correct source time; the code accepts exactly up to the limit + and rejects one frame past it. Two tests now pin that boundary. + REJECTED +
+
+ PR #2 + Sandboxed Python could delete the source media. The + argument guard could not catch it, and the container is no defence, since the workspace is + exactly what it is meant to reach. The tool now asks for approval first. + FIXED +
+
+ +

+ Qodo publishes no check run and no approving review, so GitHub's own auto-merge has nothing to + gate on. The workflow in this repo is that missing gate: it reads only the structured counter + chips, never the prose, because Qodo quotes findings verbatim and a pull request can put the + words "no issues found" into its own diff. It binds the verdict to the reviewed commit and + merges with --match-head-commit, so a push racing the merge is + rejected rather than slipped in. +

+
+ + +
+

And the video you
just watched?

+

It cut itself.

+ +
+ + + +
+
+ + + + diff --git a/site/vercel.json b/site/vercel.json new file mode 100644 index 0000000..92bf592 --- /dev/null +++ b/site/vercel.json @@ -0,0 +1,13 @@ +{ + "$schema": "https://openapi.vercel.sh/vercel.json", + "cleanUrls": true, + "headers": [ + { + "source": "/(.*)", + "headers": [ + { "key": "X-Content-Type-Options", "value": "nosniff" }, + { "key": "Referrer-Policy", "value": "strict-origin-when-cross-origin" } + ] + } + ] +} From 6ce0200b4ba702a9ef714e2678636c903ad70311 Mon Sep 17 00:00:00 2001 From: Deon Menezes Date: Sat, 29 Aug 2026 17:39:33 -0700 Subject: [PATCH 2/5] Give the GitHub links their mark The navbar CTA said "GitHub" in text, which is the one link on the page people scan for by icon rather than by reading. The mark is defined once as an SVG symbol and used in all three places the link appears: the navbar, the outro button, and the footer, so they read as one system instead of three unrelated links. The violet button picks up an inset top highlight and a soft cast of its own colour, which is what separates a button from a coloured rectangle at this size. Below 480px the label is clipped rather than hidden, so the button becomes a square mark and still announces "GitHub" to a screen reader. --- site/index.html | 48 ++++++++++++++++++++++++++++++++++++++++-------- 1 file changed, 40 insertions(+), 8 deletions(-) diff --git a/site/index.html b/site/index.html index a61f4f6..42e10f6 100644 --- a/site/index.html +++ b/site/index.html @@ -67,11 +67,19 @@ } .slate__link:hover{color:var(--fg); background:var(--panel)} .slate__cta{ - font-family:var(--mono); font-size:12px; text-decoration:none; - color:#fff; background:var(--violet); padding:8px 14px; border-radius:6px; - font-weight:500; transition:filter .18s, transform .18s; + display:inline-flex; align-items:center; gap:8px; flex:none; + font-family:var(--mono); font-size:12px; text-decoration:none; line-height:1; + color:#fff; background:var(--violet); padding:9px 14px; border-radius:7px; + font-weight:500; border:1px solid rgba(255,255,255,.16); + box-shadow:inset 0 1px 0 rgba(255,255,255,.18), 0 6px 18px -10px var(--violet); + transition:filter .18s var(--ease), transform .18s var(--ease), box-shadow .18s var(--ease); } -.slate__cta:hover{filter:brightness(1.12); transform:translateY(-1px)} +.slate__cta:hover{ + filter:brightness(1.1); transform:translateY(-1px); + box-shadow:inset 0 1px 0 rgba(255,255,255,.24), 0 10px 24px -10px var(--violet); +} +.slate__cta:active{transform:translateY(0)} +.gh{display:block; flex:none} /* ── rail: a scrub ruler down the left edge, playhead follows scroll ─── */ .rail{ @@ -301,7 +309,12 @@ font-family:var(--mono); font-size:13px; text-decoration:none; padding:13px 22px; border-radius:7px; transition:filter .18s, transform .18s, border-color .18s, color .18s; } -.btn--primary{background:var(--violet); color:#fff; font-weight:500} +.btn{display:inline-flex; align-items:center; gap:9px; line-height:1} +.btn--primary{ + background:var(--violet); color:#fff; font-weight:500; + border:1px solid rgba(255,255,255,.16); + box-shadow:inset 0 1px 0 rgba(255,255,255,.18), 0 8px 24px -12px var(--violet); +} .btn--primary:hover{filter:brightness(1.12); transform:translateY(-1px)} .btn--ghost{border:1px solid var(--line-2); color:var(--fg-dim)} .btn--ghost:hover{color:var(--fg); border-color:var(--fg-faint)} @@ -312,6 +325,7 @@ font-family:var(--mono); font-size:11.5px; color:var(--fg-faint); } footer a{color:var(--fg-dim); text-decoration:none} +.foot-gh{display:inline-flex; align-items:center; gap:7px; transition:color .18s} footer a:hover{color:var(--fg)} footer .sp{flex:1} @@ -334,9 +348,14 @@ .ledger__row{grid-template-columns:1fr; gap:7px} .ledger__verdict{text-align:left} .slate__links .slate__link{display:none} + .slate__cta{padding:9px 11px} .msg{max-width:96%} } @media (min-width:901px){.rail-mobile{display:none}} +@media (max-width:480px){ + .slate__cta-label{position:absolute; width:1px; height:1px; overflow:hidden; clip-path:inset(50%)} + .slate__cta{padding:9px} +} @media (prefers-reduced-motion:reduce){ *,*::before,*::after{animation-duration:.001ms!important; animation-iteration-count:1!important; transition-duration:.001ms!important; scroll-behavior:auto!important} @@ -346,6 +365,10 @@ + +
GitHub +
@@ -86,11 +86,16 @@ harness runs the agent loop; EditAI supplies the domain. browser harness domain ┌──────────────┐ HTTP ┌──────────────┐ MCP ┌──────────────────┐ │ editor UI │◄─────────►│ TrueForge │◄────────►│ @editai/agent │ -│ (apps/web) │ + SSE │ agent loop │ │ 16 timeline │ -│ │ │ approvals │ │ tools │ -│ timeline ◄──┼───────────┼──────────────┼── SSE ───┤ project store │ -└──────────────┘ └──────┬───────┘ └──────────────────┘ - │ MCP +│ (apps/web) │ + SSE │ agent loop │ │ 19 timeline │ +│ decode │ │ approvals │ │ tools │ +│ composite │ │ │ │ project store │ +│ encode │ │ │ │ media on disk │ +│ timeline ◄──┼───────────┼──────────────┼── SSE ───┤ render queue │ +└──────┬───────┘ └──────┬───────┘ └────────┬─────────┘ + │ │ MCP │ + │ media bytes + rendered mp4 (HTTP) │ + └───────────────────────────────────────────────────────┘ + │ ┌─────────────┼──────────────┐ ▼ ▼ ▼ ┌──────────────┐ ┌─────────┐ ┌──────────────┐ @@ -100,6 +105,19 @@ harness runs the agent loop; EditAI supplies the domain. └──────────────┘ └─────────┘ └──────────────┘ ``` +**The editor is the renderer.** Decoding and encoding happen in the browser through +[WebCodecs](https://developer.mozilla.org/en-US/docs/Web/API/WebCodecs_API), wrapped by +[mediabunny](https://mediabunny.dev): the agent queues a render, the editor claims it, composites +every frame onto a canvas, muxes it, and posts the file back. The agent then sees a real path and a +real byte count, which is what lets it check its own work. + +The engine follows [OpenCut](https://github.com/opencut-app/opencut-classic), whose renderer this is +ported from: the same frame cache (a forward iterator with a prefetched next frame, falling back to a +real seek only when the target is behind the decoder or too far ahead of it) and the same +`Output`/`CanvasSource`/`AudioBufferSource` muxing. The one deliberate departure is compositing in +Canvas2D rather than OpenCut's wgpu compositor: EditAI stacks video, text and audio with no effects +or masks, which 2D covers exactly, and it drops a wasm dependency. Effects would need the real thing. + The editor never calls a model. It creates a session, streams turn events, renders tool calls and sub-agent threads, and answers the harness when it pauses. The timeline redraws from the agent server's event stream, so an edit the agent makes shows up in the UI as it happens. @@ -115,29 +133,32 @@ server's event stream, so an edit the agent makes shows up in the UI as it happe | Sandboxed execution | Two layers: the harness sandbox (Daytona) for general code, and `packages/ffmpeg-sandbox` for media work | | Domain know-how | `skills/video-editing/SKILL.md`, loaded by the agent whenever a task touches ffmpeg | -## The 16 timeline tools +## The 19 timeline tools -The timeline is exposed to the agent as **16 [MCP](https://modelcontextprotocol.io) tools**, not as +The timeline is exposed to the agent as **19 [MCP](https://modelcontextprotocol.io) tools**, not as a prompt describing a timeline. The agent calls real functions against real state. | Tool | Kind | Arguments | What it does | | --- | --- | --- | --- | | `get_project` | read | | Tracks, clips, media metadata, exports. Call it first; ids change after edits. | +| `list_media` | read | | Imported files with their measured duration, resolution and frame rate, and whether the bytes are on disk. | | `list_changes` | read | `limit` | Recent edits, oldest first. | | `transcribe_clip` | read | `clip_id` | Speech inside one clip, as timed segments in timeline seconds. | -| `find_silences` | read | `min_duration`, `track_id` | Silent ranges on a track. Preview only. | -| `detect_beats` | read | `track_id` | Beat timestamps derived from the track's tempo. | +| `find_silences` | read | `min_duration`, `track_id` | Silent ranges measured from the decoded audio. Preview only. | +| `detect_beats` | read | `track_id` | Beat timestamps from the tempo estimated off the track's onsets. | | `split_clip` | write | `clip_id`, `at` | Cuts a clip in two. Returns both halves. | | `trim_clip` | write | `clip_id`, `start?`, `end?` | New in/out points, keeping media in sync via `sourceOffset`. | | `move_clip` | write | `clip_id`, `start?`, `track_id?` | New start time, or another track of the same kind. | | `set_volume` | write | `clip_id`, `volume` | Clip volume, 0 to 100. | +| `add_clip` | write | `name`, `track_id`, `start`, `duration?`, `source_offset?` | Places imported media on a video or audio track. | | `add_text` | write | `text`, `start`, `duration`, `track_id` | A title or caption on a text track. | | `add_captions` | write | `segments[]`, `track_label` | Timed captions on the captions track, creating it if needed. | | `undo` | write | | Reverts the most recent change. | | `delete_clip` | **destructive** | `clip_id` | Removes a clip, leaving a gap. | | `ripple_delete` | **destructive** | `start`, `end` | Removes a range from every track and closes the gap. | | `remove_silences` | **destructive** | `min_duration`, `track_id` | Ripple-deletes every silence over the threshold. | -| `export_project` | **approval** | `format`, `resolution` | Renders the timeline to a file. | +| `export_project` | **approval** | `format`, `resolution` | Queues a real render. Refuses if any clip's media is missing. | +| `get_export` | read | `export_id` | Render status: pending, rendering with progress, done with the file and its byte size, or failed with the error. | Every tool validates its input with zod and returns errors to the model **as data**, so a stale clip id becomes a correction the agent recovers from rather than a failed turn. @@ -247,10 +268,21 @@ ANTHROPIC_API_KEY=sk-... bun run setup cd ../.. && bun run dev:web # http://localhost:5173 ``` -Then ask for an edit: "Remove the silences", "Caption every video clip". +Then bring in footage. Drop any video or audio file onto the editor, or use **Import** in the Media +panel: the browser measures it with WebCodecs, uploads the bytes to the agent, and analyzes the audio +so silences, the waveform and tempo come from your file. No footage handy? + +```bash +cd apps/agent && bun scripts/make-samples.ts # writes data/samples with ffmpeg +``` + +Then ask for an edit: "Remove the silences", "Caption every video clip", "Export it at 1080p". The +export lands in `apps/agent/data/exports` as a real mp4, and the Export button turns into a download +link once it does. Without the harness running, the editor still loads with a sample timeline; the assistant panel -says it is offline. `setup` is idempotent, so rerun it after changing `agent.json` or adding a key. +says it is offline. That sample timeline names media it has no bytes for, so the preview says so +and export refuses until real files are imported. `setup` is idempotent, so rerun it after changing `agent.json` or adding a key. ### Environment @@ -277,10 +309,18 @@ documented set. ``` apps/ web/ the editor: timeline, preview, assistant panel + src/engine/ decode, composite, encode: the renderer, ported from OpenCut + media.ts probing and upload over range-requested HTTP + video-cache.ts the frame cache: forward iterator, prefetch, seek fallback + audio.ts decode, timeline mixdown, silence/peak/tempo analysis + compositor.ts one frame of the timeline, drawn in Canvas2D + exporter.ts mediabunny mux: CanvasSource + AudioBufferSource agent/ MCP server exposing the timeline, plus the agent definition - src/tools.ts the 16 tools - src/project.ts the timeline model: split, trim, ripple delete, captions + src/tools.ts the 19 tools + src/project.ts the timeline model: split, trim, ripple delete, captions, renders + src/media.ts media names, mime types, streamed uploads scripts/setup.ts registers models, connectors, sandbox and the agent + scripts/make-samples.ts generates real sample footage with ffmpeg packages/ ffmpeg-sandbox/ containerised ffmpeg/ffprobe/python MCP server skills/ @@ -304,11 +344,13 @@ tests/ the Qodo verdict parser's fixtures and tests ## Tests -**31 tests**, all passing. +**58 tests**, all passing. | Suite | Count | Covers | | --- | --- | --- | -| `apps/agent/test/project.test.ts` | 17 | Split and trim invariants, ripple-delete arithmetic across tracks, silence removal, transcript windowing, caption merging, undo, and the clip-id and trim-bound regressions Qodo surfaced. | +| `apps/agent/test/project.test.ts` | 30 | Split and trim invariants, ripple-delete arithmetic across tracks, silence removal, transcript windowing, caption merging, undo, media registration and clip placement bounds, and the render lifecycle: one claim per job, and a finished render that a late progress or failure report cannot reopen. | +| `apps/web/src/engine/audio.test.ts` | 10 | Silence detection against injected gaps and sub-threshold room tone, the peak envelope, and tempo recovered from a synthetic click track. | +| `apps/web/src/engine/compositor.test.ts` | 8 | Which clips are live at a time, source-time mapping through `sourceOffset`, and which clips reach the audio mix. | | `apps/web/.../transcript.test.ts` | 6 | Transcript windowing and caption rendering in the editor. | | `tests/test_qodo_verdict.py` | 8 | The merge gate's verdict parser, against the real Qodo comment bodies that broke it. | @@ -397,6 +439,8 @@ rejected: - [TanStack Start](https://tanstack.com/start) + React 19, Vite, Tailwind CSS v4, shadcn/ui - [TrueForge](https://trueforge.dev) agent harness, [MCP](https://modelcontextprotocol.io) tools +- [mediabunny](https://mediabunny.dev) over WebCodecs for decode, mux and encode, with the frame + cache and export pipeline ported from [OpenCut](https://github.com/opencut-app/opencut-classic) - Cloudflare Workers (via Wrangler), Bun + Turborepo monorepo - Landing page: hand-written static HTML in `site/`, deployed to Vercel @@ -404,10 +448,16 @@ rejected: Worth stating plainly, because the demo does not make them obvious: -- `export_project` writes a JSON description of the render rather than encoding video. Wiring it to - ffmpeg is the obvious next step and does not change the agent-facing contract. -- Transcripts, silences and tempo come from media metadata in the sample project rather than from - running ASR and onset detection over real files. +- **Transcripts are still fixtures.** Silences, the waveform and tempo are now measured from the + decoded audio, but `transcribe_clip` reads transcript segments stored on the media rather than + running ASR. Wiring a real recognizer in is the next gap to close. +- **Rendering needs the editor open.** The agent queues a render and the browser performs it, so + `export_project` from a headless session waits for a page to claim the job. A server-side ffmpeg + worker consuming the same queue would fix it without changing any tool signature. +- **Compositing is Canvas2D.** Video, text and audio composite correctly; effects, transitions, + masks and blend modes have nowhere to live. Those need OpenCut's wgpu compositor, not this one. +- **Codecs are the browser's.** Import refuses anything Chrome cannot decode, and mp4 audio falls + back to Opus where AAC encoding is unavailable. - Motion graphics have no dedicated timeline tool yet. Today they go through the ffmpeg sandbox or an attached MCP server. diff --git a/apps/agent/.gitignore b/apps/agent/.gitignore index 236730b..1169630 100644 --- a/apps/agent/.gitignore +++ b/apps/agent/.gitignore @@ -1,2 +1,6 @@ +# Runtime state: the working project, imported media, renders and generated samples. +# A clone should start from the seed, not from somebody's timeline. data/project.json data/exports/ +data/media/ +data/samples/ diff --git a/apps/agent/data/project.json b/apps/agent/data/project.json deleted file mode 100644 index c0b503e..0000000 --- a/apps/agent/data/project.json +++ /dev/null @@ -1,337 +0,0 @@ -{ - "name": "Untitled project", - "fps": 30, - "duration": 21.1, - "tracks": [ - { - "id": "v1", - "label": "V1", - "kind": "video" - }, - { - "id": "t1", - "label": "T1", - "kind": "text" - }, - { - "id": "a1", - "label": "A1", - "kind": "audio" - }, - { - "id": "a2", - "label": "A2", - "kind": "audio" - }, - { - "id": "t2", - "label": "T2", - "kind": "text" - } - ], - "clips": [ - { - "id": "c1", - "name": "intro.mp4", - "kind": "video", - "trackId": "v1", - "start": 0, - "duration": 3.2, - "sourceOffset": 0 - }, - { - "id": "c1r", - "name": "intro.mp4", - "kind": "video", - "trackId": "v1", - "start": 3.2, - "duration": 0.9, - "sourceOffset": 4.1 - }, - { - "id": "c2", - "name": "b-roll.mp4", - "kind": "video", - "trackId": "v1", - "start": 4.1, - "duration": 4.6, - "sourceOffset": 0 - }, - { - "id": "c2r", - "name": "b-roll.mp4", - "kind": "video", - "trackId": "v1", - "start": 8.7, - "duration": 0.6, - "sourceOffset": 5.4 - }, - { - "id": "c3", - "name": "talking-head.mp4", - "kind": "video", - "trackId": "v1", - "start": 9.3, - "duration": 5.8, - "sourceOffset": 0 - }, - { - "id": "c3r", - "name": "talking-head.mp4", - "kind": "video", - "trackId": "v1", - "start": 15.1, - "duration": 6, - "sourceOffset": 7 - }, - { - "id": "c4", - "name": "Hook line", - "kind": "text", - "trackId": "t1", - "start": 0.5, - "duration": 2.7, - "sourceOffset": 0 - }, - { - "id": "c5", - "name": "Subscribe", - "kind": "text", - "trackId": "t1", - "start": 17.1, - "duration": 4, - "sourceOffset": 0 - }, - { - "id": "c6", - "name": "voiceover.wav", - "kind": "audio", - "trackId": "a1", - "start": 0, - "duration": 3.2, - "sourceOffset": 0, - "volume": 100 - }, - { - "id": "c6r", - "name": "voiceover.wav", - "kind": "audio", - "trackId": "a1", - "start": 3.2, - "duration": 5.5, - "sourceOffset": 4.1, - "volume": 100 - }, - { - "id": "c6r", - "name": "voiceover.wav", - "kind": "audio", - "trackId": "a1", - "start": 8.7, - "duration": 6.4, - "sourceOffset": 10.4, - "volume": 100 - }, - { - "id": "c6r", - "name": "voiceover.wav", - "kind": "audio", - "trackId": "a1", - "start": 15.1, - "duration": 6, - "sourceOffset": 18, - "volume": 100 - }, - { - "id": "c7", - "name": "music.mp3", - "kind": "audio", - "trackId": "a2", - "start": 0, - "duration": 3.2, - "sourceOffset": 0, - "volume": 35 - }, - { - "id": "c7r", - "name": "music.mp3", - "kind": "audio", - "trackId": "a2", - "start": 3.2, - "duration": 5.5, - "sourceOffset": 4.1, - "volume": 35 - }, - { - "id": "c7r", - "name": "music.mp3", - "kind": "audio", - "trackId": "a2", - "start": 8.7, - "duration": 6.4, - "sourceOffset": 10.4, - "volume": 35 - }, - { - "id": "c7r", - "name": "music.mp3", - "kind": "audio", - "trackId": "a2", - "start": 15.1, - "duration": 6, - "sourceOffset": 18, - "volume": 35 - }, - { - "id": "c8", - "name": "Most editors waste hours on cuts a machine should make.", - "kind": "text", - "trackId": "t2", - "start": 0.4, - "duration": 2.7, - "sourceOffset": 0 - }, - { - "id": "c9", - "name": "EditAI reads your timeline and does the boring parts.", - "kind": "text", - "trackId": "t2", - "start": 3.3, - "duration": 0.8, - "sourceOffset": 0 - }, - { - "id": "c10", - "name": "EditAI reads your timeline and does the boring parts.", - "kind": "text", - "trackId": "t2", - "start": 4.1, - "duration": 2.5, - "sourceOffset": 0 - }, - { - "id": "c11", - "name": "Silences, captions, pacing.", - "kind": "text", - "trackId": "t2", - "start": 6.7, - "duration": 1.9, - "sourceOffset": 0 - }, - { - "id": "c12", - "name": "You describe the change, it edits, you approve.", - "kind": "text", - "trackId": "t2", - "start": 8.8, - "duration": 0.5, - "sourceOffset": 0 - }, - { - "id": "c13", - "name": "You describe the change, it edits, you approve.", - "kind": "text", - "trackId": "t2", - "start": 9.3, - "duration": 3, - "sourceOffset": 0 - }, - { - "id": "c14", - "name": "Every destructive step waits for a human.", - "kind": "text", - "trackId": "t2", - "start": 12.4, - "duration": 2.6, - "sourceOffset": 0 - }, - { - "id": "c15", - "name": "It runs on any model and any tools you plug in.", - "kind": "text", - "trackId": "t2", - "start": 15.2, - "duration": 2.9, - "sourceOffset": 0 - }, - { - "id": "c16", - "name": "Subscribe if you want to see where this goes.", - "kind": "text", - "trackId": "t2", - "start": 18.3, - "duration": 2.4, - "sourceOffset": 0 - } - ], - "media": { - "intro.mp4": { - "duration": 5 - }, - "b-roll.mp4": { - "duration": 6 - }, - "talking-head.mp4": { - "duration": 13 - }, - "voiceover.wav": { - "duration": 24, - "silences": [ - { - "start": 3.2, - "end": 4.1 - }, - { - "start": 9.6, - "end": 10.4 - }, - { - "start": 16.8, - "end": 18 - } - ], - "transcript": [ - { - "start": 0.4, - "end": 3.1, - "text": "Most editors waste hours on cuts a machine should make." - }, - { - "start": 4.2, - "end": 7.5, - "text": "EditAI reads your timeline and does the boring parts." - }, - { - "start": 7.6, - "end": 9.5, - "text": "Silences, captions, pacing." - }, - { - "start": 10.5, - "end": 14, - "text": "You describe the change, it edits, you approve." - }, - { - "start": 14.1, - "end": 16.7, - "text": "Every destructive step waits for a human." - }, - { - "start": 18.1, - "end": 21, - "text": "It runs on any model and any tools you plug in." - }, - { - "start": 21.2, - "end": 23.6, - "text": "Subscribe if you want to see where this goes." - } - ] - }, - "music.mp3": { - "duration": 24, - "bpm": 120 - } - }, - "exports": [] -} \ No newline at end of file diff --git a/apps/agent/scripts/make-samples.ts b/apps/agent/scripts/make-samples.ts new file mode 100644 index 0000000..aa3c27b --- /dev/null +++ b/apps/agent/scripts/make-samples.ts @@ -0,0 +1,81 @@ +/** + * Generate real sample footage with ffmpeg, for trying EditAI without your own media. + * + * These are genuine encoded files, not fixtures: the editor decodes them, the analyzer + * measures them, and the exporter re-encodes them. The voiceover has real gaps at known + * times so silence detection has something true to find. + * + * bun scripts/make-samples.ts [outDir] + */ +import { spawn } from "node:child_process" +import { existsSync, mkdirSync } from "node:fs" +import { dirname, join } from "node:path" +import { fileURLToPath } from "node:url" + +const here = dirname(fileURLToPath(import.meta.url)) +const outDir = process.argv[2] ?? join(here, "..", "data", "samples") + +/** Gaps the voiceover really contains, so `find_silences` can be checked against the truth. */ +export const SILENCES = [ + { start: 3.2, end: 4.1 }, + { start: 9.6, end: 10.4 }, + { start: 16.8, end: 18.0 }, +] + +const gate = SILENCES.map((s) => `between(t,${s.start},${s.end})`).join("+") + +const CLIPS: { file: string; args: string[] }[] = [ + { + file: "intro.mp4", + args: ["-f", "lavfi", "-i", "testsrc2=size=1280x720:rate=30:duration=5", "-pix_fmt", "yuv420p", "-c:v", "libx264", "-preset", "veryfast"], + }, + { + file: "b-roll.mp4", + args: ["-f", "lavfi", "-i", "smptebars=size=1280x720:rate=30:duration=6", "-pix_fmt", "yuv420p", "-c:v", "libx264", "-preset", "veryfast"], + }, + { + file: "talking-head.mp4", + args: [ + "-f", "lavfi", "-i", "gradients=size=1280x720:rate=30:duration=13:c0=0x2b2440:c1=0x0e0e10", + "-pix_fmt", "yuv420p", "-c:v", "libx264", "-preset", "veryfast", + ], + }, + { + file: "voiceover.wav", + args: [ + "-f", "lavfi", + "-i", `aevalsrc='if(${gate},0,0.35*sin(2*PI*210*t)*(0.55+0.45*sin(2*PI*3.1*t)))':d=24:s=48000`, + "-c:a", "pcm_s16le", + ], + }, + { + file: "music.mp3", + // 120 BPM: a decaying click every half second, so tempo estimation has a real beat. + args: ["-f", "lavfi", "-i", "aevalsrc='0.45*sin(2*PI*760*t)*exp(-26*mod(t,0.5))':d=24:s=48000", "-c:a", "libmp3lame", "-b:a", "192k"], + }, +] + +function run(args: string[]): Promise { + return new Promise((resolve, reject) => { + const child = spawn("ffmpeg", ["-y", "-hide_banner", "-loglevel", "error", ...args], { stdio: ["ignore", "ignore", "pipe"] }) + let stderr = "" + child.stderr.on("data", (d) => (stderr += d)) + child.on("error", (err) => reject(new Error(`ffmpeg could not start: ${err.message}. Install it with: brew install ffmpeg`))) + child.on("close", (code) => (code === 0 ? resolve() : reject(new Error(stderr.trim() || `ffmpeg exited with ${code}`)))) + }) +} + +if (import.meta.main) { + mkdirSync(outDir, { recursive: true }) + for (const clip of CLIPS) { + const target = join(outDir, clip.file) + if (existsSync(target)) { + console.log(` exists ${clip.file}`) + continue + } + await run([...clip.args, target]) + console.log(` wrote ${clip.file}`) + } + console.log(`\nSample media in ${outDir}`) + console.log("Drop these onto the editor to import them. The voiceover has real silences at 3.2s, 9.6s and 16.8s.") +} diff --git a/apps/agent/src/media.ts b/apps/agent/src/media.ts new file mode 100644 index 0000000..0ff388a --- /dev/null +++ b/apps/agent/src/media.ts @@ -0,0 +1,68 @@ +import { createWriteStream, existsSync, mkdirSync, renameSync, statSync, unlinkSync } from "node:fs" +import { randomUUID } from "node:crypto" +import { basename, extname, join } from "node:path" +import type { IncomingMessage } from "node:http" +import { pipeline } from "node:stream/promises" + +/** 2 GiB. Enough for real footage, small enough that a runaway upload cannot fill the disk. */ +export const MAX_UPLOAD_BYTES = 2 * 1024 * 1024 * 1024 + +const MIME: Record = { + ".mp4": "video/mp4", + ".mov": "video/quicktime", + ".webm": "video/webm", + ".mkv": "video/x-matroska", + ".m4v": "video/mp4", + ".wav": "audio/wav", + ".mp3": "audio/mpeg", + ".m4a": "audio/mp4", + ".aac": "audio/aac", + ".ogg": "audio/ogg", + ".flac": "audio/flac", +} + +export const mimeFor = (name: string) => MIME[extname(name).toLowerCase()] ?? "application/octet-stream" + +/** + * Media names come from the browser and end up as paths, so they are reduced to a bare + * file name. `basename` alone is not enough: a name is also used as a project key, and + * "..", an empty string or a leading dot would each produce a path that is not a file + * inside the media dir. + */ +export function safeMediaName(raw: string): string { + const name = basename(String(raw ?? "").trim()).replace(/[/\\]/g, "") + if (!name || name === "." || name === "..") throw new Error("Media name is not a usable file name.") + if (name.startsWith(".")) throw new Error("Media name cannot start with a dot.") + if (name.length > 200) throw new Error("Media name is too long.") + return name +} + +/** Stream a request body to disk, refusing anything over the cap. */ +export async function saveUpload(req: IncomingMessage, dir: string, name: string): Promise { + mkdirSync(dir, { recursive: true }) + const target = join(dir, name) + // Unique: two uploads of the same name must not write through one another's temp file. + const tmp = `${target}.${randomUUID()}.part` + let bytes = 0 + let tooBig = false + req.on("data", (chunk: Buffer) => { + bytes += chunk.length + if (bytes > MAX_UPLOAD_BYTES && !tooBig) { + tooBig = true + req.destroy(new Error(`Upload exceeds ${MAX_UPLOAD_BYTES} bytes.`)) + } + }) + try { + await pipeline(req, createWriteStream(tmp)) + } catch (err) { + if (existsSync(tmp)) unlinkSync(tmp) + throw err + } + if (bytes === 0) { + unlinkSync(tmp) + throw new Error("Upload was empty.") + } + // Rename last so a reader never sees a half-written file under the real name. + renameSync(tmp, target) + return statSync(target).size +} diff --git a/apps/agent/src/project.ts b/apps/agent/src/project.ts index 38c650e..d80f03e 100644 --- a/apps/agent/src/project.ts +++ b/apps/agent/src/project.ts @@ -1,5 +1,5 @@ -import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs" -import { dirname } from "node:path" +import { existsSync, mkdirSync, readFileSync, renameSync, statSync, writeFileSync } from "node:fs" +import { dirname, join } from "node:path" export type ClipKind = "video" | "text" | "audio" @@ -31,15 +31,38 @@ export type MediaInfo = { transcript?: Segment[] /** beats per minute, for music */ bpm?: number + /** Set once real bytes are on disk. File name inside the media dir. */ + file?: string + width?: number + height?: number + fps?: number + hasAudio?: boolean + sizeBytes?: number + /** Peak envelope over the whole file, 0..1, for the timeline waveform. */ + peaks?: number[] + /** Where silences/peaks came from: absent means they were never measured. */ + analyzedAt?: string } +export type ExportStatus = "pending" | "rendering" | "done" | "failed" + export type ExportRecord = { id: string format: string resolution: string + width: number + height: number + fps: number createdAt: string durationSeconds: number file: string + status: ExportStatus + /** 0..1 while rendering. */ + progress?: number + /** Real bytes on disk, only once status is "done". */ + sizeBytes?: number + completedAt?: string + error?: string } export type Project = { @@ -112,6 +135,11 @@ export function seedProject(): Project { } } +/** A project with the standard track layout and nothing on it, for starting from real footage. */ +export function emptyProject(): Project { + return { ...seedProject(), clips: [], media: {}, duration: 0, exports: [] } +} + export class ProjectStore { private project: Project private snapshots: Project[] = [] @@ -119,10 +147,21 @@ export class ProjectStore { private listeners = new Set<(p: Project, change: Change | null) => void>() revision = 0 - constructor(private file?: string) { + constructor( + private file?: string, + /** Where media bytes live. Given, the store can tell a registered file from a present one. */ + private mediaDir?: string, + ) { this.project = file && existsSync(file) ? (JSON.parse(readFileSync(file, "utf8")) as Project) : seedProject() } + /** Media is only usable once its bytes are actually on disk, not merely named. */ + hasBytes(name: string): boolean { + const info = this.project.media[name] + if (!info?.file) return false + return this.mediaDir ? existsSync(join(this.mediaDir, info.file)) : true + } + get(): Project { return structuredClone(this.project) } @@ -136,10 +175,10 @@ export class ProjectStore { return () => this.listeners.delete(fn) } - reset() { + reset(opts: { empty?: boolean } = {}) { this.snapshots = [] this.changes = [] - this.project = seedProject() + this.project = opts.empty ? emptyProject() : seedProject() this.persist(null) } @@ -404,22 +443,217 @@ export class ProjectStore { return { track, clips: this.project.clips.filter((c) => ids.includes(c.id)) } } - exportProject(format: string, resolution: string, dir: string): ExportRecord { + // ---- real media ----------------------------------------------------------- + + /** + * Record media whose bytes are on disk. Metadata is measured by the editor (WebCodecs) + * rather than guessed here, so the agent server needs no ffprobe of its own. + */ + registerMedia(name: string, info: Omit): MediaInfo { + if (!name.trim()) throw new Error("Media needs a name.") + this.commit("import_media", `Imported ${name} (${round(info.duration)}s)`, (p) => { + p.media[name] = { ...p.media[name], ...info } + }) + return this.project.media[name]! + } + + /** Attach measurements taken from the decoded audio: silences, peak envelope, tempo. */ + setMediaAnalysis(name: string, analysis: { silences?: Segment[]; peaks?: number[]; bpm?: number; transcript?: Segment[] }): MediaInfo { + const media = this.project.media[name] + if (!media) throw new Error(`No media named "${name}".`) + const counted = analysis.silences ? `${analysis.silences.length} silences` : "waveform" + this.commit("analyze_media", `Analyzed ${name}: ${counted}`, (p) => { + p.media[name] = { ...p.media[name]!, ...analysis, analyzedAt: new Date().toISOString() } + }) + return this.project.media[name]! + } + + /** Put imported media on the timeline. Unlike addTextClip this needs a real source file. */ + addClip(opts: { name: string; trackId: string; start: number; duration?: number; sourceOffset?: number }): Clip { + const track = this.track(opts.trackId) + if (track.kind === "text") throw new Error(`${track.label} is a text track; use add_text instead.`) + const media = this.project.media[opts.name] + if (!media) throw new Error(`No media named "${opts.name}". Call list_media to see what has been imported.`) + if (!this.hasBytes(opts.name)) throw new Error(`"${opts.name}" has no media on disk yet, so it cannot be placed on the timeline.`) + const sourceOffset = opts.sourceOffset ?? 0 + if (sourceOffset < 0 || sourceOffset >= media.duration) { + throw new Error(`sourceOffset ${sourceOffset}s is outside ${opts.name} (0s to ${media.duration}s).`) + } + const duration = opts.duration ?? media.duration - sourceOffset + if (duration <= 0) throw new Error("Duration must be positive.") + if (sourceOffset + duration > media.duration + 1e-6) { + throw new Error(`${opts.name} only has ${round(media.duration - sourceOffset)}s left after a ${sourceOffset}s offset.`) + } + if (opts.start < 0) throw new Error("Clips cannot start before 0s.") + let id = "" + this.commit("add_clip", `Added ${opts.name} at ${round(opts.start)}s on ${track.label}`, (p) => { + id = this.newId("c") + p.clips.push({ + id, + name: opts.name, + kind: track.kind, + trackId: track.id, + start: opts.start, + duration, + sourceOffset, + ...(track.kind === "audio" ? { volume: 100 } : {}), + }) + }) + return this.clip(id) + } + + // ---- export --------------------------------------------------------------- + + /** + * Queue a render. The encode happens in the editor, which owns the decoders, so this + * only creates the job; the file appears when the editor posts the bytes back. + */ + requestExport(format: string, resolution: string, dir: string): ExportRecord { + const missing = this.missingMedia() + if (missing.length > 0) { + throw new Error( + `Cannot render: ${missing.join(", ")} ${missing.length === 1 ? "has" : "have"} no media on disk. ` + + `Import real footage first, or remove those clips.`, + ) + } + const { width, height } = resolutionToSize(resolution) + const used = new Set(this.project.exports.map((e) => e.id)) + let n = this.project.exports.length + 1 + while (used.has(`exp${n}`)) n++ const rec: ExportRecord = { - id: `exp${this.project.exports.length + 1}`, + id: `exp${n}`, format, resolution, + width, + height, + fps: this.project.fps, createdAt: new Date().toISOString(), durationSeconds: this.project.duration, - file: `${dir}/${this.project.name.replace(/\s+/g, "-").toLowerCase()}-${resolution}.${format}`, + file: join(dir, `${this.project.name.replace(/\s+/g, "-").toLowerCase()}-${resolution}-exp${n}.${format}`), + status: "pending", + progress: 0, } - // Written before the commit: commit notifies subscribers synchronously, and a client that - // reacts to the new export record must not find the file missing. mkdirSync(dir, { recursive: true }) - writeFileSync(rec.file, JSON.stringify({ export: rec, project: this.project }, null, 2)) - this.commit("export_project", `Exported ${resolution} ${format}`, (p) => { + this.commit("export_project", `Queued a ${resolution} ${format} render`, (p) => { p.exports.push(rec) }) return rec } + + /** Clips whose source media has no bytes on disk. Nothing can be rendered from those. */ + missingMedia(): string[] { + const names = new Set() + for (const c of this.project.clips) { + if (c.kind === "text") continue + if (!this.hasBytes(c.name)) names.add(c.name) + } + return [...names] + } + + getExport(id: string): ExportRecord { + const rec = this.project.exports.find((e) => e.id === id) + if (!rec) throw new Error(`No export with id "${id}".`) + return structuredClone(rec) + } + + /** The oldest render the editor has not picked up yet. */ + pendingExport(): ExportRecord | null { + return structuredClone(this.project.exports.find((e) => e.status === "pending") ?? null) + } + + private updateExport(id: string, fn: (rec: ExportRecord) => void, op: string, summary: string) { + if (!this.project.exports.some((e) => e.id === id)) throw new Error(`No export with id "${id}".`) + this.commit(op, summary, (p) => { + fn(p.exports.find((e) => e.id === id)!) + }) + return this.getExport(id) + } + + /** + * Take a queued render. Returns null if it is already claimed, which is how two editors + * open on the same project avoid both encoding it: only one claim can win. + */ + claimExport(id: string): ExportRecord | null { + const rec = this.project.exports.find((e) => e.id === id) + if (!rec || rec.status !== "pending") return null + return this.updateExport( + id, + (r) => { + r.status = "rendering" + r.progress = 0 + }, + "export_claimed", + `Started rendering ${id}`, + ) + } + + setExportProgress(id: string, progress: number) { + // A finished render must not be reopened by a straggling progress report. + const current = this.getExport(id) + if (current.status !== "rendering") return current + const clamped = Math.max(0, Math.min(1, progress)) + return this.updateExport( + id, + (rec) => { + rec.progress = clamped + }, + "export_progress", + `Rendering ${id}: ${Math.round(clamped * 100)}%`, + ) + } + + /** + * Finish a render whose bytes are already on disk at `uploadedPath`. + * + * A path rather than a buffer: a 4K render is gigabytes, and reading it into the server only + * to write it out again would be the one place this design needs the whole file in memory. + */ + completeExport(id: string, uploadedPath: string) { + const rec = this.getExport(id) + if (!existsSync(uploadedPath)) throw new Error(`No uploaded file at ${uploadedPath}.`) + // Moved before the commit: commit notifies subscribers synchronously, and a client that + // reacts to the finished export must not find the file missing. + mkdirSync(dirname(rec.file), { recursive: true }) + renameSync(uploadedPath, rec.file) + const sizeBytes = statSync(rec.file).size + return this.updateExport( + id, + (r) => { + r.status = "done" + r.progress = 1 + r.sizeBytes = sizeBytes + r.completedAt = new Date().toISOString() + delete r.error + }, + "export_done", + `Rendered ${rec.resolution} ${rec.format} (${(sizeBytes / 1e6).toFixed(1)} MB)`, + ) + } + + failExport(id: string, error: string) { + // Same reason: a late failure from a losing worker must not bury a finished file. + const current = this.getExport(id) + if (current.status === "done") return current + return this.updateExport( + id, + (rec) => { + rec.status = "failed" + rec.error = error + }, + "export_failed", + `Render ${id} failed: ${error}`, + ) + } +} + +/** Render sizes are 16:9, matching the editor's preview. */ +export function resolutionToSize(resolution: string): { width: number; height: number } { + switch (resolution) { + case "720p": + return { width: 1280, height: 720 } + case "4k": + return { width: 3840, height: 2160 } + default: + return { width: 1920, height: 1080 } + } } diff --git a/apps/agent/src/server.ts b/apps/agent/src/server.ts index e2cecf2..f76ab30 100644 --- a/apps/agent/src/server.ts +++ b/apps/agent/src/server.ts @@ -1,22 +1,32 @@ import { StreamableHTTPServerTransport } from "@modelcontextprotocol/sdk/server/streamableHttp.js" import cors from "cors" import express from "express" +import { createReadStream, existsSync, statSync } from "node:fs" import { dirname, join } from "node:path" import { fileURLToPath } from "node:url" +import { mimeFor, safeMediaName, saveUpload } from "./media.ts" import { ProjectStore } from "./project.ts" import { buildServer } from "./tools.ts" const here = dirname(fileURLToPath(import.meta.url)) const DATA_DIR = process.env.EDITAI_DATA_DIR ?? join(here, "..", "data") +const MEDIA_DIR = join(DATA_DIR, "media") +const EXPORTS_DIR = join(DATA_DIR, "exports") const PORT = Number(process.env.EDITAI_AGENT_PORT ?? 8941) -const store = new ProjectStore(join(DATA_DIR, "project.json")) +const store = new ProjectStore(join(DATA_DIR, "project.json"), MEDIA_DIR) const app = express() app.use(cors({ origin: true })) -app.use(express.json({ limit: "2mb" })) + +/** Uploads stream straight to disk, so they must not be buffered by the JSON parser first. */ +const isUpload = (url: string) => /^\/(media\/[^/]+|exports\/[^/]+\/file)$/.test(url.split("?")[0]!) +app.use((req, res, next) => (req.method === "POST" && isUpload(req.url) ? next() : express.json({ limit: "8mb" })(req, res, next))) + +const fail = (res: express.Response, status: number, err: unknown) => + res.status(status).json({ error: err instanceof Error ? err.message : String(err) }) app.get("/", (_req, res) => { - res.type("text/plain").send("EditAI agent server. MCP at POST /mcp. Project at GET /project, live at GET /events.") + res.type("text/plain").send("EditAI agent server. MCP at POST /mcp. Project at GET /project, live at GET /events, media at GET /media.") }) app.get("/healthz", (_req, res) => res.json({ ok: true, revision: store.revision })) @@ -25,11 +35,148 @@ app.get("/project", (_req, res) => { res.json({ project: store.get(), revision: store.revision, changes: store.listChanges(20) }) }) -app.post("/project/reset", (_req, res) => { - store.reset() +app.post("/project/reset", (req, res) => { + store.reset({ empty: req.body?.empty === true }) res.json({ project: store.get(), revision: store.revision }) }) +// ---- media ---------------------------------------------------------------- + +app.get("/media", (_req, res) => res.json({ media: store.get().media, dir: MEDIA_DIR })) + +/** + * The editor measures media with WebCodecs and posts the bytes here, so the server needs + * no decoder of its own. Metadata rides in the query string; the body is the file. + */ +app.post("/media/:name", async (req, res) => { + try { + const name = safeMediaName(req.params.name) + const q = req.query as Record + const duration = Number(q.duration) + if (!Number.isFinite(duration) || duration <= 0) throw new Error("A positive ?duration= in seconds is required.") + const sizeBytes = await saveUpload(req, MEDIA_DIR, name) + const num = (v: string | undefined) => (v === undefined || v === "" ? undefined : Number(v)) + const info = store.registerMedia(name, { + duration, + file: name, + sizeBytes, + width: num(q.width), + height: num(q.height), + fps: num(q.fps), + hasAudio: q.hasAudio === undefined ? undefined : q.hasAudio === "true", + }) + res.json({ media: info, name }) + } catch (err) { + fail(res, 400, err) + } +}) + +app.post("/media/:name/analysis", (req, res) => { + try { + res.json({ media: store.setMediaAnalysis(safeMediaName(req.params.name), req.body ?? {}) }) + } catch (err) { + fail(res, 400, err) + } +}) + +app.get("/media/:name", (req, res) => { + try { + const name = safeMediaName(req.params.name) + const file = join(MEDIA_DIR, name) + if (!existsSync(file)) return fail(res, 404, new Error(`No media file named "${name}".`)) + const { size } = statSync(file) + res.setHeader("Content-Type", mimeFor(name)) + res.setHeader("Accept-Ranges", "bytes") + // Range support keeps a