diff --git a/electron/ai-edition/agent-tools.test.ts b/electron/ai-edition/agent-tools.test.ts index 4bad12b91..936a5f9e6 100644 --- a/electron/ai-edition/agent-tools.test.ts +++ b/electron/ai-edition/agent-tools.test.ts @@ -183,6 +183,7 @@ describe("the mutating-tool table", () => { "removeClip", "removeModifier", "removeTrim", + "removeFillerWords", "replaceTimeline", "setAnnotation", "setAudio", @@ -2361,6 +2362,181 @@ describe("getTranscriptWords", () => { }); }); +describe("removeFillerWords", () => { + function repeatedWordDocument(): AxcutDocument { + const base = fixtureDocument(); + return { + ...base, + transcripts: [ + { + ...base.transcripts[0], + words: [ + { id: "filler", segmentId: "seg_1", startSec: 1, endSec: 1.2, text: "like" }, + { id: "meaningful", segmentId: "seg_1", startSec: 3, endSec: 3.2, text: "like" }, + ], + }, + ], + }; + } + + it("cuts only the selected occurrence and reports the trim actually stored", () => { + const before = repeatedWordDocument(); + const read = run(before, "getTranscriptWords", {}); + expect(JSON.parse(read.resultJson).words.map((word: { id: string }) => word.id)).toEqual([ + "filler", + "meaningful", + ]); + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(true); + const next = result.document as AxcutDocument; + const trim = next.timeline.trimRanges.at(-1); + expect(trim).toMatchObject({ assetId: "asset_1", clipId: "clip_1", origin: "agent" }); + expect(trim?.endSec).toBeLessThan(3); // the second "like" still plays + expect(next.transcripts).toEqual(before.transcripts); + expect(next.timeline.clips).toEqual(before.timeline.clips); + const payload = JSON.parse(result.resultJson); + expect(payload).toMatchObject({ + requested: 1, + removedCount: 1, + removed: [ + { + wordId: "filler", + text: "like", + assetId: "asset_1", + clipId: "clip_1", + wordStartSec: 1, + wordEndSec: 1.2, + trimRangeId: trim?.id, + startSec: trim?.startSec, + endSec: trim?.endSec, + }, + ], + }); + expect(next.timeline.trimRanges).toHaveLength(before.timeline.trimRanges.length + 1); + }); + + it("preserves addTrim warnings when removing a filler cuts a zoom transition", () => { + const before = repeatedWordDocument(); + before.timeline.trimRanges = []; + const zoomed = run(before, "addZoom", { startSec: 2, endSec: 3, depth: 3 }).document; + expect(zoomed).toBeDefined(); + if (!zoomed) throw new Error("Zoom fixture failed"); + const directTrim = run(zoomed, "addTrim", { + assetId: "asset_1", + clipId: "clip_1", + startSec: 1, + endSec: 1.2, + }); + const expected = JSON.parse(directTrim.resultJson).cutTransitions; + expect(expected.length).toBeGreaterThan(0); + const removed = run(zoomed, "removeFillerWords", { assetId: "asset_1", wordIds: ["filler"] }); + expect(removed.ok).toBe(true); + expect(JSON.parse(removed.resultJson).cutTransitions).toEqual(expected); + }); + + it("refuses an invalid ID atomically, including when another ID is valid", () => { + const before = repeatedWordDocument(); + const result = run(before, "removeFillerWords", { wordIds: ["filler", "missing"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("Unknown transcript word ID: missing"); + expect(before.timeline.trimRanges).toHaveLength(1); + }); + + it("resolves a word ID to its own asset and clip, not the primary asset", () => { + const before = repeatedWordDocument(); + before.assets.push({ ...before.assets[0], id: "asset_2" }); + before.timeline.clips.push({ + ...before.timeline.clips[0], + id: "clip_3", + assetId: "asset_2", + }); + before.transcripts.push({ + ...before.transcripts[0], + assetId: "asset_2", + words: [{ id: "other_filler", segmentId: "seg_1", startSec: 4, endSec: 4.2, text: "um" }], + }); + const result = run(before, "removeFillerWords", { wordIds: ["other_filler"] }); + expect(result.ok).toBe(true); + expect(result.document?.timeline.trimRanges.at(-1)).toMatchObject({ + assetId: "asset_2", + clipId: "clip_3", + }); + expect(JSON.parse(result.resultJson).removed[0]).toMatchObject({ + assetId: "asset_2", + clipId: "clip_3", + }); + }); + + it("qualifies normally numbered word IDs by the asset read from the transcript", () => { + const before = repeatedWordDocument(); + before.transcripts[0].words[0].id = "word_1"; + before.assets.push({ ...before.assets[0], id: "asset_2" }); + before.timeline.clips.push({ ...before.timeline.clips[0], id: "clip_3", assetId: "asset_2" }); + before.transcripts.push({ + ...before.transcripts[0], + assetId: "asset_2", + words: [{ id: "word_1", segmentId: "seg_1", startSec: 4, endSec: 4.2, text: "um" }], + }); + const read = JSON.parse(run(before, "getTranscriptWords", { assetId: "asset_2" }).resultJson); + expect(read.words[0].id).toBe("word_1"); + const result = run(before, "removeFillerWords", { + assetId: "asset_2", + wordIds: [read.words[0].id], + }); + expect(result.ok).toBe(true); + expect(result.document?.timeline.trimRanges.at(-1)).toMatchObject({ + assetId: "asset_2", + clipId: "clip_3", + }); + expect( + result.document?.timeline.trimRanges.filter((trim) => trim.assetId === "asset_1"), + ).toEqual(before.timeline.trimRanges); + expect(result.document?.transcripts).toEqual(before.transcripts); + const ambiguous = run(before, "removeFillerWords", { wordIds: ["word_1"] }); + expect(ambiguous.ok).toBe(false); + expect(ambiguous.document).toBeUndefined(); + const unknownAsset = run(before, "removeFillerWords", { + assetId: "missing_asset", + wordIds: ["word_1"], + }); + expect(unknownAsset.ok).toBe(false); + expect(unknownAsset.document).toBeUndefined(); + }); + + it("refuses invalid timestamps without guessing a span", () => { + const before = repeatedWordDocument(); + before.transcripts[0].words[0].endSec = Number.NaN; + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("invalid source timestamps"); + }); + + it("refuses a word covered by two clips", () => { + const before = repeatedWordDocument(); + before.timeline.clips.push({ + ...before.timeline.clips[0], + id: "duplicate_clip", + }); + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("exactly one clip; found 2"); + }); + + it("honours project edit consent", () => { + const result = executeAgentTool( + repeatedWordDocument(), + "removeFillerWords", + JSON.stringify({ wordIds: ["filler"] }), + { editsAllowed: false }, + ); + expect(result.document).toBeUndefined(); + expect(JSON.parse(result.resultJson).code).toBe("consent_required"); + }); +}); + describe("setWordText", () => { it("changes the text and leaves the timeline alone", () => { const before = documentWithWords(); diff --git a/electron/ai-edition/agent-tools.ts b/electron/ai-edition/agent-tools.ts index b8ffb68a4..df08c897a 100644 --- a/electron/ai-edition/agent-tools.ts +++ b/electron/ai-edition/agent-tools.ts @@ -424,6 +424,11 @@ export const addTrimsArgs = z.object({ ranges: z.array(z.union([addTrimArgs, z.unknown()])).min(1), }); +export const removeFillerWordsArgs = z.object({ + assetId: z.string().min(1).optional(), + wordIds: z.array(z.string().min(1)).min(1), +}); + export const setTrimArgs = z.object({ trimRangeId: z.string().min(1), startSec: secondsSchema, @@ -619,6 +624,7 @@ export const OPENSCREEN_TOOL_NAMES = [ "getTranscriptWords", "getCursorTrack", "setWordText", + "removeFillerWords", "addTrim", "addTrims", "setTrim", @@ -690,6 +696,7 @@ export const MUTATING_TOOL_NAMES: ReadonlySet = new Set([ // Writes the transcript, not the timeline — but it writes the document, so it is a // consented edit like any other. "setWordText", + "removeFillerWords", "addTrim", "addTrims", "addZooms", @@ -1485,6 +1492,157 @@ export function executeAgentTool( }; } + case "removeFillerWords": { + const parsed = removeFillerWordsArgs.safeParse(args); + if (!parsed.success) return failure(parsed.error.message); + const { assetId: selectedAssetId, wordIds } = parsed.data; + if (new Set(wordIds).size !== wordIds.length) { + return failure("Duplicate word IDs in one request. Nothing was modified."); + } + + // Resolve every ID before writing anything. Word IDs, rather than guessed + // timestamps, designate which occurrence of a repeated word is speech to cut. + const transcripts = [...document.transcripts]; + if ( + document.transcript && + !transcripts.some((transcript) => transcript.assetId === document.transcript?.assetId) + ) { + transcripts.push(document.transcript); + } + const selected: Array<{ + wordId: string; + text: string; + assetId: string; + clipId: string; + startSec: number; + endSec: number; + }> = []; + for (const wordId of wordIds) { + const matches = transcripts + .filter( + (transcript) => selectedAssetId === undefined || transcript.assetId === selectedAssetId, + ) + .flatMap((transcript) => + transcript.words + .filter((word) => word.id === wordId) + .map((word) => ({ assetId: transcript.assetId, word })), + ); + if (matches.length !== 1) { + return failure( + matches.length === 0 + ? `Unknown transcript word ID: ${wordId}. Call getTranscriptWords to read the IDs. Nothing was modified.` + : `Word ID ${wordId} appears more than once in the transcripts. Pass the assetId used for getTranscriptWords. Nothing was modified.`, + ); + } + const { assetId, word } = matches[0]; + const asset = document.assets.find((candidate) => candidate.id === assetId); + if (!asset || isGeneratedAssetId(assetId) || word.source === "synth") { + return failure(`Word ${wordId} has no recorded source asset. Nothing was modified.`); + } + const { startSec, endSec } = word; + if ( + !Number.isFinite(startSec) || + !Number.isFinite(endSec) || + startSec < 0 || + endSec <= startSec || + (asset.durationSec !== undefined && + (!Number.isFinite(asset.durationSec) || endSec > asset.durationSec)) + ) { + return failure(`Word ${wordId} has invalid source timestamps. Nothing was modified.`); + } + const clips = document.timeline.clips.filter((clip) => { + const clipEnd = clip.sourceEndSec ?? asset.durationSec; + return ( + clip.assetId === assetId && + clipEnd !== undefined && + Number.isFinite(clipEnd) && + startSec >= clip.sourceStartSec && + endSec <= clipEnd + ); + }); + if (clips.length !== 1) { + return failure( + `Word ${wordId} must be fully contained in exactly one clip; found ${clips.length}. Nothing was modified.`, + ); + } + const clip = clips[0]; + if ( + document.timeline.trimRanges.some( + (trim) => + trimAppliesToClip(trim, clip) && startSec < trim.endSec && endSec > trim.startSec, + ) + ) { + return failure(`Word ${wordId} is already cut by a trim. Nothing was modified.`); + } + selected.push({ wordId, text: word.text, assetId, clipId: clip.id, startSec, endSec }); + } + + let current = document; + const removed: Array> = []; + for (const word of selected) { + const execution = executeAgentTool( + current, + "addTrim", + JSON.stringify({ + assetId: word.assetId, + clipId: word.clipId, + startSec: word.startSec, + endSec: word.endSec, + reason: "filler word", + }), + options, + ); + if (!execution.ok || !execution.document) { + return failure( + `Could not cut word ${word.wordId}: ${execution.resultJson}. Nothing was modified.`, + ); + } + const { trimRangeId } = JSON.parse(execution.resultJson) as { trimRangeId: string }; + const trim = execution.document.timeline.trimRanges.find( + (range) => range.id === trimRangeId, + ); + const clip = current.timeline.clips.find((candidate) => candidate.id === word.clipId); + const asset = current.assets.find((candidate) => candidate.id === word.assetId); + const clipEnd = clip?.sourceEndSec ?? asset?.durationSec; + if ( + !trim || + !clip || + clipEnd === undefined || + trim.clipId !== word.clipId || + !Number.isFinite(trim.startSec) || + !Number.isFinite(trim.endSec) || + trim.startSec < clip.sourceStartSec || + trim.endSec > clipEnd || + trim.endSec <= trim.startSec + ) { + return failure(`Cut for word ${word.wordId} left its clip. Nothing was modified.`); + } + current = execution.document; + removed.push({ + wordId: word.wordId, + text: word.text, + assetId: word.assetId, + clipId: word.clipId, + wordStartSec: word.startSec, + wordEndSec: word.endSec, + trimRangeId, + startSec: trim.startSec, + endSec: trim.endSec, + }); + } + return { + ok: true, + document: current, + resultJson: JSON.stringify({ + requested: wordIds.length, + removedCount: removed.length, + removed, + ...cutTransitionsReport(document, current), + }), + summary: `removed ${removed.length} filler word${removed.length === 1 ? "" : "s"}`, + }; + } + case "addTrim": { const parsed = addTrimArgs.safeParse(args); if (!parsed.success) return failure(parsed.error.message); diff --git a/electron/ai-edition/deep-agent/service.test.ts b/electron/ai-edition/deep-agent/service.test.ts index 78326fdfc..e2bb9e16d 100644 --- a/electron/ai-edition/deep-agent/service.test.ts +++ b/electron/ai-edition/deep-agent/service.test.ts @@ -60,6 +60,7 @@ const ARGS: Record = { getTranscriptWords: {}, getCursorTrack: {}, setWordText: { wordId: "word_1", text: "Hullo" }, + removeFillerWords: { wordIds: ["word_1"] }, addTrim: { startSec: 1, endSec: 2 }, addTrims: { ranges: [{ startSec: 1, endSec: 2 }] }, setTrim: { trimRangeId: "trim_1", startSec: 1, endSec: 2 }, diff --git a/electron/ai-edition/deep-agent/service.ts b/electron/ai-edition/deep-agent/service.ts index b7924e870..b91066b1a 100644 --- a/electron/ai-edition/deep-agent/service.ts +++ b/electron/ai-edition/deep-agent/service.ts @@ -44,6 +44,7 @@ import { isMutatingTool, moveClipArgs, removeClipArgs, + removeFillerWordsArgs, removeModifierArgs, removeTrimArgs, replaceTimelineArgs, @@ -95,7 +96,7 @@ const CONSENT_PROMPT_BLOCK = [ "", "PROJECT EDITS ARE CURRENTLY DISABLED by the user, who asked to be consulted before the timeline changes.", "- Read freely: getCurrentDocument and getTranscript work as usual.", - "- Do NOT call any tool that writes (addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.", + "- Do NOT call any tool that writes (removeFillerWords, addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.", "- Instead: say precisely what you would change — which tool, which times, which ids — and ask the user to confirm. Be specific enough that they can say yes to it.", "- Never state or imply that an edit was applied. If the user confirms and you are still refused, tell them the 'Project edits' setting in Settings → AI has to be re-enabled first.", ].join("\n"); @@ -119,6 +120,7 @@ const BASE_SYSTEM_PROMPT = [ // other than English. Say what the tool does; let the model do the matching. "How the tools map to intent — pick the most specific one, and prefer the smallest edit that satisfies the request:", "- Silences, pauses and dead stretches are removed as trims INSIDE the placed clip. Send them together with addTrims once you know the ranges; addTrim is for a single cut or a correction. The placed clip stays the canonical cut; it is not rebuilt to drop them.", + "- Only when the user explicitly asks to remove filler words: read getTranscriptWords, decide which occurrences are fillers from their surrounding speech, then pass that assetId and ONLY those word IDs to removeFillerWords. Word IDs restart for each asset, so keep the read's assetId when selecting occurrences. The tool resolves their exact source and clip, makes trims, and reports the actual cuts. Do not remove every occurrence of a word just because its text matches a filler elsewhere.", "- Changing where a clip starts or ends within its source is setClipRange — the clip's in/out, distinct from a trim.", `- addZoom takes a virtual-timeline span (depth is an ordinal 1–6 selecting from a fixed table — ${ZOOM_DEPTH_LEGEND} — never a multiplier; focus in 0–1 frame fractions). addSpeed changes pacing over a span. addAnnotation puts text on screen. addCameraFullscreen enlarges the webcam, and only does something where assets[].hasCameraTrack is true.`, "- A zoom animates in just before its span and out just after it (zoomInFromSec, zoomOutUntilSec in getCurrentDocument). Leave those windows untrimmed, or the export jumps at the cut.", @@ -156,9 +158,12 @@ export const TOOL_DESCRIPTIONS: Record = { getCursorTrack: "Read the recorded pointer track for an asset: where the cursor was over time, downsampled to a readable rate. Each point carries atSec (the asset's own source clock), virtualSec (the same instant on the edited timeline — the coordinate addZoom takes, null when no clip carries it), cx/cy as 0–1 fractions of the frame, and `shape`, an index into the pointer bitmaps the recording used (equal values are the same pointer; a change means the pointer changed, e.g. arrow to text caret). Points that are not plain moves carry `kind`; points a trim cuts out of playback carry `trimmed`. These are real samples, not a summary — reading what the pointer was doing is yours. Omit assetId for the primary asset. It answers `available:false` in two DIFFERENT ways you must not confuse: reason 'no-sidecar' means this asset was checked and genuinely has no telemetry, while reason 'unavailable' means it could not be read from here.", getTranscriptWords: - 'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText takes; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.', + 'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText and removeFillerWords take; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.', setWordText: "Correct ONE word's text, by the id getTranscriptWords returns. This changes the TRANSCRIPT and nothing else: the captions follow it, the film is untouched and no audio is cut. Use it when the transcriber misheard something — a name, a technical term — and the user asks for it to read correctly. Passing an empty string BLANKS the word: it keeps its place in the media but leaves the captions, which is how a junk token like \"(inaudible)\" is removed without cutting the speech around it. Writing the transcriber's own text back clears the correction. This is NOT how you make a spoken word go away — that removes only the label and leaves the film saying it; use addTrim, which cuts the audio with it.", + removeFillerWords: + "Remove only the spoken filler-word occurrences selected by `wordIds` from getTranscriptWords. Pass the same `assetId` as that read: word IDs are numbered independently in each asset. Omit assetId only when the IDs are unique across all transcripts. Use ONLY when the user explicitly asks for filler-word cleanup. Read words in context and choose IDs yourself; the executor does not classify speech or remove every matching text. It resolves each ID to one recorded asset, valid source timestamps, and exactly one fully covering clip, then adds non-destructive trims. If any ID or target is invalid or ambiguous, nothing changes. The result lists each word and the ACTUAL trim span (which addTrim may shape), asset, clip and trim ID; report only those results. " + + CUT_TRANSITIONS_RESULT, addTrim: `Add ONE trim range: a cut of a span inside a clip (this source-time span will not be played or exported) that does NOT split the clip. Times are in seconds of the asset's source time. This is the preferred (and for 'remove silences' requests, the only) way to handle silences; it preserves the user's placed clips and only adds a cut. When you have several cuts to make, use addTrims and send them together — this one is for a single cut or a later correction. A cut belongs to ONE clip: \`clipId\` is inferred when a single clip covers the range, but when several clips draw on the same asset over it the call FAILS and lists them — pass the \`clipId\` you mean (ids come from getCurrentDocument). ${ZOOM_TRANSITIONS_NOTE} ${CUT_TRANSITIONS_RESULT}`, addTrims: `Add MANY trim ranges in one call: \`ranges\` is a list, each entry taking exactly the fields addTrim takes. Use this whenever you have more than one cut to make — 'remove the silences' on a half-hour recording is hundreds of cuts, and sending them one at a time costs one round trip each. Each range stands or falls ALONE: one that cannot be placed is refused by itself and listed in \`refused\` with its index and the reason, while every other range is still applied. Nothing is rolled back, so a single bad bound never costs you the rest. The result leads with requested / appliedCount / refusedCount so you can see a partial outcome without re-reading the document — report what was refused rather than claiming the whole list landed. ${ZOOM_TRANSITIONS_NOTE} Each applied entry carries the same \`cutTransitions\` warning addTrim reports.`, setTrim: `Move or resize an existing trim range by id. Times are source-time seconds. The cut follows to whichever clip the new range lands in, when that clip is unambiguous. ${ZOOM_TRANSITIONS_NOTE} ${CUT_TRANSITIONS_RESULT}`, @@ -367,6 +372,7 @@ export const TOOL_ARG_SCHEMAS: ReadonlyArray = [ ["getTranscriptWords", getTranscriptWordsArgs], ["getCursorTrack", getCursorTrackArgs], ["setWordText", setWordTextArgs], + ["removeFillerWords", removeFillerWordsArgs], ["addTrim", addTrimArgs], ["addTrims", addTrimsArgs], ["setTrim", setTrimArgs], diff --git a/electron/mcp/openscreen-mcp-server.test.ts b/electron/mcp/openscreen-mcp-server.test.ts index a19bf535d..607b88055 100644 --- a/electron/mcp/openscreen-mcp-server.test.ts +++ b/electron/mcp/openscreen-mcp-server.test.ts @@ -83,6 +83,34 @@ class FakeEditor implements McpDocumentHost { } } +function fillerEditor(): FakeEditor { + const editor = new FakeEditor(); + editor.document = documentSchema.parse({ + ...editor.document, + transcripts: [ + { + assetId: "asset_1", + language: "en", + segments: [ + { + id: "seg_1", + kind: "speech", + startSec: 0, + endSec: 5, + text: "like I like", + wordIds: ["filler", "meaningful"], + }, + ], + words: [ + { id: "filler", segmentId: "seg_1", startSec: 1, endSec: 1.2, text: "like" }, + { id: "meaningful", segmentId: "seg_1", startSec: 3, endSec: 3.2, text: "like" }, + ], + }, + ], + }); + return editor; +} + let running: RunningMcpServer | null = null; let client: Client | null = null; @@ -135,6 +163,10 @@ describe("the MCP tool surface", () => { expect(Object.keys(addTrim?.inputSchema.properties ?? {})).toEqual( expect.arrayContaining(["startSec", "endSec"]), ); + const removeFillerWords = tools.find((t) => t.name === "removeFillerWords"); + expect(Object.keys(removeFillerWords?.inputSchema.properties ?? {})).toEqual( + expect.arrayContaining(["assetId", "wordIds"]), + ); }); it("marks reads read-only and deletions destructive", async () => { @@ -155,6 +187,59 @@ describe("the MCP tool surface", () => { }); describe("calling a tool", () => { + it("cuts the selected filler occurrence through the shared tool surface", async () => { + const editor = fillerEditor(); + const mcp = await connect(editor); + const result = await mcp.callTool({ + name: "removeFillerWords", + arguments: { wordIds: ["filler"] }, + }); + expect(result.isError).toBeFalsy(); + expect(editor.applied).toHaveLength(1); + const trim = editor.document?.timeline.trimRanges.at(-1); + expect(trim).toMatchObject({ assetId: "asset_1", clipId: "clip_1", origin: "agent" }); + expect(trim?.endSec).toBeLessThan(3); + expect(JSON.parse(resultText(result)).removed[0]).toMatchObject({ + wordId: "filler", + trimRangeId: trim?.id, + startSec: trim?.startSec, + endSec: trim?.endSec, + }); + }); + + it("keeps the asset qualifier when MCP words have duplicate IDs across recordings", async () => { + const editor = fillerEditor(); + const before = editor.document as AxcutDocument; + before.assets.push({ ...before.assets[0], id: "asset_2" }); + before.timeline.clips.push({ ...before.timeline.clips[0], id: "clip_2", assetId: "asset_2" }); + before.transcripts.push({ ...before.transcripts[0], assetId: "asset_2" }); + const mcp = await connect(editor); + const result = await mcp.callTool({ + name: "removeFillerWords", + arguments: { assetId: "asset_2", wordIds: ["filler"] }, + }); + expect(result.isError).toBeFalsy(); + expect(editor.applied).toHaveLength(1); + expect(editor.document?.timeline.trimRanges).toHaveLength(1); + expect(editor.document?.timeline.trimRanges[0]).toMatchObject({ + assetId: "asset_2", + clipId: "clip_2", + }); + }); + + it("refuses filler removal when MCP project edits are disabled", async () => { + const editor = fillerEditor(); + const before = editor.document; + const mcp = await connect(editor, { editsAllowed: false }); + const result = await mcp.callTool({ + name: "removeFillerWords", + arguments: { wordIds: ["filler"] }, + }); + expect(result.isError).toBe(true); + expect(resultText(result)).toMatch(/consent_required/); + expect(editor.document).toBe(before); + expect(editor.applied).toHaveLength(0); + }); it("reads the live editor document", async () => { const mcp = await connect(new FakeEditor()); const result = await mcp.callTool({ name: "getCurrentDocument", arguments: {} });