From cd7e904348a66268914ead38628fd75cb38212c2 Mon Sep 17 00:00:00 2001 From: auberginewly <178153638+auberginewly@users.noreply.github.com> Date: Sun, 27 Sep 2026 14:29:22 +0800 Subject: [PATCH] feat: remove selected filler words by transcript ID --- electron/ai-edition/agent-tools.test.ts | 121 ++++++++++++++ electron/ai-edition/agent-tools.ts | 152 ++++++++++++++++++ .../ai-edition/deep-agent/service.test.ts | 1 + electron/ai-edition/deep-agent/service.ts | 9 +- 4 files changed, 281 insertions(+), 2 deletions(-) diff --git a/electron/ai-edition/agent-tools.test.ts b/electron/ai-edition/agent-tools.test.ts index d8ffa054d..f43945bb7 100644 --- a/electron/ai-edition/agent-tools.test.ts +++ b/electron/ai-edition/agent-tools.test.ts @@ -183,6 +183,7 @@ describe("the mutating-tool table", () => { "removeClip", "removeModifier", "removeTrim", + "removeFillerWords", "replaceTimeline", "setAnnotation", "setAudio", @@ -2331,6 +2332,126 @@ describe("getTranscriptWords", () => { }); }); +describe("removeFillerWords", () => { + function repeatedWordDocument(): AxcutDocument { + const base = fixtureDocument(); + return { + ...base, + transcripts: [ + { + ...base.transcripts[0], + words: [ + { id: "filler", segmentId: "seg_1", startSec: 1, endSec: 1.2, text: "like" }, + { id: "meaningful", segmentId: "seg_1", startSec: 3, endSec: 3.2, text: "like" }, + ], + }, + ], + }; + } + + it("cuts only the selected occurrence and reports the trim actually stored", () => { + const before = repeatedWordDocument(); + const read = run(before, "getTranscriptWords", {}); + expect(JSON.parse(read.resultJson).words.map((word: { id: string }) => word.id)).toEqual([ + "filler", + "meaningful", + ]); + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(true); + const next = result.document as AxcutDocument; + const trim = next.timeline.trimRanges.at(-1); + expect(trim).toMatchObject({ assetId: "asset_1", clipId: "clip_1", origin: "agent" }); + expect(trim?.endSec).toBeLessThan(3); // the second "like" still plays + expect(next.transcripts).toEqual(before.transcripts); + expect(next.timeline.clips).toEqual(before.timeline.clips); + const payload = JSON.parse(result.resultJson); + expect(payload).toMatchObject({ + requested: 1, + removedCount: 1, + removed: [ + { + wordId: "filler", + text: "like", + assetId: "asset_1", + clipId: "clip_1", + wordStartSec: 1, + wordEndSec: 1.2, + trimRangeId: trim?.id, + startSec: trim?.startSec, + endSec: trim?.endSec, + }, + ], + }); + expect(next.timeline.trimRanges).toHaveLength(before.timeline.trimRanges.length + 1); + }); + + it("refuses an invalid ID atomically, including when another ID is valid", () => { + const before = repeatedWordDocument(); + const result = run(before, "removeFillerWords", { wordIds: ["filler", "missing"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("Unknown transcript word ID: missing"); + expect(before.timeline.trimRanges).toHaveLength(1); + }); + + it("resolves a word ID to its own asset and clip, not the primary asset", () => { + const before = repeatedWordDocument(); + before.assets.push({ ...before.assets[0], id: "asset_2" }); + before.timeline.clips.push({ + ...before.timeline.clips[0], + id: "clip_3", + assetId: "asset_2", + }); + before.transcripts.push({ + ...before.transcripts[0], + assetId: "asset_2", + words: [{ id: "other_filler", segmentId: "seg_1", startSec: 4, endSec: 4.2, text: "um" }], + }); + const result = run(before, "removeFillerWords", { wordIds: ["other_filler"] }); + expect(result.ok).toBe(true); + expect(result.document?.timeline.trimRanges.at(-1)).toMatchObject({ + assetId: "asset_2", + clipId: "clip_3", + }); + expect(JSON.parse(result.resultJson).removed[0]).toMatchObject({ + assetId: "asset_2", + clipId: "clip_3", + }); + }); + + it("refuses invalid timestamps without guessing a span", () => { + const before = repeatedWordDocument(); + before.transcripts[0].words[0].endSec = Number.NaN; + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("invalid source timestamps"); + }); + + it("refuses a word covered by two clips", () => { + const before = repeatedWordDocument(); + before.timeline.clips.push({ + ...before.timeline.clips[0], + id: "duplicate_clip", + }); + const result = run(before, "removeFillerWords", { wordIds: ["filler"] }); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + expect(result.resultJson).toContain("exactly one clip; found 2"); + }); + + it("honours project edit consent", () => { + const result = executeAgentTool( + repeatedWordDocument(), + "removeFillerWords", + JSON.stringify({ wordIds: ["filler"] }), + { editsAllowed: false }, + ); + expect(result.document).toBeUndefined(); + expect(JSON.parse(result.resultJson).code).toBe("consent_required"); + }); +}); + describe("setWordText", () => { it("changes the text and leaves the timeline alone", () => { const before = documentWithWords(); diff --git a/electron/ai-edition/agent-tools.ts b/electron/ai-edition/agent-tools.ts index bb25b8448..0efc30e28 100644 --- a/electron/ai-edition/agent-tools.ts +++ b/electron/ai-edition/agent-tools.ts @@ -414,6 +414,10 @@ export const addTrimsArgs = z.object({ ranges: z.array(z.union([addTrimArgs, z.unknown()])).min(1), }); +export const removeFillerWordsArgs = z.object({ + wordIds: z.array(z.string().min(1)).min(1), +}); + export const setTrimArgs = z.object({ trimRangeId: z.string().min(1), startSec: secondsSchema, @@ -599,6 +603,7 @@ export const OPENSCREEN_TOOL_NAMES = [ "getTranscriptWords", "getCursorTrack", "setWordText", + "removeFillerWords", "addTrim", "addTrims", "setTrim", @@ -670,6 +675,7 @@ export const MUTATING_TOOL_NAMES: ReadonlySet = new Set([ // Writes the transcript, not the timeline — but it writes the document, so it is a // consented edit like any other. "setWordText", + "removeFillerWords", "addTrim", "addTrims", "addZooms", @@ -1406,6 +1412,152 @@ export function executeAgentTool( }; } + case "removeFillerWords": { + const parsed = removeFillerWordsArgs.safeParse(args); + if (!parsed.success) return failure(parsed.error.message); + const { wordIds } = parsed.data; + if (new Set(wordIds).size !== wordIds.length) { + return failure("Duplicate word IDs in one request. Nothing was modified."); + } + + // Resolve every ID before writing anything. Word IDs, rather than guessed + // timestamps, designate which occurrence of a repeated word is speech to cut. + const transcripts = [...document.transcripts]; + if ( + document.transcript && + !transcripts.some((transcript) => transcript.assetId === document.transcript?.assetId) + ) { + transcripts.push(document.transcript); + } + const selected: Array<{ + wordId: string; + text: string; + assetId: string; + clipId: string; + startSec: number; + endSec: number; + }> = []; + for (const wordId of wordIds) { + const matches = transcripts.flatMap((transcript) => + transcript.words + .filter((word) => word.id === wordId) + .map((word) => ({ assetId: transcript.assetId, word })), + ); + if (matches.length !== 1) { + return failure( + matches.length === 0 + ? `Unknown transcript word ID: ${wordId}. Call getTranscriptWords to read the IDs. Nothing was modified.` + : `Word ID ${wordId} appears more than once in the transcripts. Nothing was modified.`, + ); + } + const { assetId, word } = matches[0]; + const asset = document.assets.find((candidate) => candidate.id === assetId); + if (!asset || isGeneratedAssetId(assetId) || word.source === "synth") { + return failure(`Word ${wordId} has no recorded source asset. Nothing was modified.`); + } + const { startSec, endSec } = word; + if ( + !Number.isFinite(startSec) || + !Number.isFinite(endSec) || + startSec < 0 || + endSec <= startSec || + (asset.durationSec !== undefined && + (!Number.isFinite(asset.durationSec) || endSec > asset.durationSec)) + ) { + return failure(`Word ${wordId} has invalid source timestamps. Nothing was modified.`); + } + const clips = document.timeline.clips.filter((clip) => { + const clipEnd = clip.sourceEndSec ?? asset.durationSec; + return ( + clip.assetId === assetId && + clipEnd !== undefined && + Number.isFinite(clipEnd) && + startSec >= clip.sourceStartSec && + endSec <= clipEnd + ); + }); + if (clips.length !== 1) { + return failure( + `Word ${wordId} must be fully contained in exactly one clip; found ${clips.length}. Nothing was modified.`, + ); + } + const clip = clips[0]; + if ( + document.timeline.trimRanges.some( + (trim) => + trimAppliesToClip(trim, clip) && startSec < trim.endSec && endSec > trim.startSec, + ) + ) { + return failure(`Word ${wordId} is already cut by a trim. Nothing was modified.`); + } + selected.push({ wordId, text: word.text, assetId, clipId: clip.id, startSec, endSec }); + } + + let current = document; + const removed: Array> = []; + for (const word of selected) { + const execution = executeAgentTool( + current, + "addTrim", + JSON.stringify({ + assetId: word.assetId, + clipId: word.clipId, + startSec: word.startSec, + endSec: word.endSec, + reason: "filler word", + }), + options, + ); + if (!execution.ok || !execution.document) { + return failure( + `Could not cut word ${word.wordId}: ${execution.resultJson}. Nothing was modified.`, + ); + } + const { trimRangeId } = JSON.parse(execution.resultJson) as { trimRangeId: string }; + const trim = execution.document.timeline.trimRanges.find( + (range) => range.id === trimRangeId, + ); + const clip = current.timeline.clips.find((candidate) => candidate.id === word.clipId); + const asset = current.assets.find((candidate) => candidate.id === word.assetId); + const clipEnd = clip?.sourceEndSec ?? asset?.durationSec; + if ( + !trim || + !clip || + clipEnd === undefined || + trim.clipId !== word.clipId || + !Number.isFinite(trim.startSec) || + !Number.isFinite(trim.endSec) || + trim.startSec < clip.sourceStartSec || + trim.endSec > clipEnd || + trim.endSec <= trim.startSec + ) { + return failure(`Cut for word ${word.wordId} left its clip. Nothing was modified.`); + } + current = execution.document; + removed.push({ + wordId: word.wordId, + text: word.text, + assetId: word.assetId, + clipId: word.clipId, + wordStartSec: word.startSec, + wordEndSec: word.endSec, + trimRangeId, + startSec: trim.startSec, + endSec: trim.endSec, + }); + } + return { + ok: true, + document: current, + resultJson: JSON.stringify({ + requested: wordIds.length, + removedCount: removed.length, + removed, + }), + summary: `removed ${removed.length} filler word${removed.length === 1 ? "" : "s"}`, + }; + } + case "addTrim": { const parsed = addTrimArgs.safeParse(args); if (!parsed.success) return failure(parsed.error.message); diff --git a/electron/ai-edition/deep-agent/service.test.ts b/electron/ai-edition/deep-agent/service.test.ts index 5aebb9a33..44683969e 100644 --- a/electron/ai-edition/deep-agent/service.test.ts +++ b/electron/ai-edition/deep-agent/service.test.ts @@ -60,6 +60,7 @@ const ARGS: Record = { getTranscriptWords: {}, getCursorTrack: {}, setWordText: { wordId: "word_1", text: "Hullo" }, + removeFillerWords: { wordIds: ["word_1"] }, addTrim: { startSec: 1, endSec: 2 }, addTrims: { ranges: [{ startSec: 1, endSec: 2 }] }, setTrim: { trimRangeId: "trim_1", startSec: 1, endSec: 2 }, diff --git a/electron/ai-edition/deep-agent/service.ts b/electron/ai-edition/deep-agent/service.ts index 125354d6a..8bc806664 100644 --- a/electron/ai-edition/deep-agent/service.ts +++ b/electron/ai-edition/deep-agent/service.ts @@ -42,6 +42,7 @@ import { isMutatingTool, moveClipArgs, removeClipArgs, + removeFillerWordsArgs, removeModifierArgs, removeTrimArgs, replaceTimelineArgs, @@ -93,7 +94,7 @@ const CONSENT_PROMPT_BLOCK = [ "", "PROJECT EDITS ARE CURRENTLY DISABLED by the user, who asked to be consulted before the timeline changes.", "- Read freely: getCurrentDocument and getTranscript work as usual.", - "- Do NOT call any tool that writes (addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.", + "- Do NOT call any tool that writes (removeFillerWords, addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.", "- Instead: say precisely what you would change — which tool, which times, which ids — and ask the user to confirm. Be specific enough that they can say yes to it.", "- Never state or imply that an edit was applied. If the user confirms and you are still refused, tell them the 'Project edits' setting in Settings → AI has to be re-enabled first.", ].join("\n"); @@ -117,6 +118,7 @@ const BASE_SYSTEM_PROMPT = [ // other than English. Say what the tool does; let the model do the matching. "How the tools map to intent — pick the most specific one, and prefer the smallest edit that satisfies the request:", "- Silences, pauses and dead stretches are removed as trims INSIDE the placed clip. Send them together with addTrims once you know the ranges; addTrim is for a single cut or a correction. The placed clip stays the canonical cut; it is not rebuilt to drop them.", + "- Only when the user explicitly asks to remove filler words: read getTranscriptWords, decide which occurrences are fillers from their surrounding speech, then pass ONLY those word IDs to removeFillerWords. The tool resolves their exact source and clip, makes trims, and reports the actual cuts. Do not remove every occurrence of a word just because its text matches a filler elsewhere.", "- Changing where a clip starts or ends within its source is setClipRange — the clip's in/out, distinct from a trim.", `- addZoom takes a virtual-timeline span (depth is an ordinal 1–6 selecting from a fixed table — ${ZOOM_DEPTH_LEGEND} — never a multiplier; focus in 0–1 frame fractions). addSpeed changes pacing over a span. addAnnotation puts text on screen. addCameraFullscreen enlarges the webcam, and only does something where assets[].hasCameraTrack is true.`, "- addAudio lays an imported voiceover or music file over a span. It plays an asset the project already has (kind 'audio'); importing or recording one is the editor's job, not a tool you have — so when the project has none, say so rather than naming an id that does not exist.", @@ -149,9 +151,11 @@ export const TOOL_DESCRIPTIONS: Record = { getCursorTrack: "Read the recorded pointer track for an asset: where the cursor was over time, downsampled to a readable rate. Each point carries atSec (the asset's own source clock), virtualSec (the same instant on the edited timeline — the coordinate addZoom takes, null when no clip carries it), cx/cy as 0–1 fractions of the frame, and `shape`, an index into the pointer bitmaps the recording used (equal values are the same pointer; a change means the pointer changed, e.g. arrow to text caret). Points that are not plain moves carry `kind`; points a trim cuts out of playback carry `trimmed`. These are real samples, not a summary — reading what the pointer was doing is yours. Omit assetId for the primary asset. It answers `available:false` in two DIFFERENT ways you must not confuse: reason 'no-sidecar' means this asset was checked and genuinely has no telemetry, while reason 'unavailable' means it could not be read from here.", getTranscriptWords: - 'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText takes; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.', + 'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText and removeFillerWords take; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.', setWordText: "Correct ONE word's text, by the id getTranscriptWords returns. This changes the TRANSCRIPT and nothing else: the captions follow it, the film is untouched and no audio is cut. Use it when the transcriber misheard something — a name, a technical term — and the user asks for it to read correctly. Passing an empty string BLANKS the word: it keeps its place in the media but leaves the captions, which is how a junk token like \"(inaudible)\" is removed without cutting the speech around it. Writing the transcriber's own text back clears the correction. This is NOT how you make a spoken word go away — that removes only the label and leaves the film saying it; use addTrim, which cuts the audio with it.", + removeFillerWords: + "Remove only the spoken filler-word occurrences selected by `wordIds` from getTranscriptWords. Use ONLY when the user explicitly asks for filler-word cleanup. Read words in context and choose IDs yourself; the executor does not classify speech or remove every matching text. It resolves each ID to one recorded asset, valid source timestamps, and exactly one fully covering clip, then adds non-destructive trims. If any ID or target is invalid or ambiguous, nothing changes. The result lists each word and the ACTUAL trim span (which addTrim may shape), asset, clip and trim ID; report only those results.", addTrim: "Add ONE trim range: a cut of a span inside a clip (this source-time span will not be played or exported) that does NOT split the clip. Times are in seconds of the asset's source time. This is the preferred (and for 'remove silences' requests, the only) way to handle silences; it preserves the user's placed clips and only adds a cut. When you have several cuts to make, use addTrims and send them together — this one is for a single cut or a later correction. A cut belongs to ONE clip: `clipId` is inferred when a single clip covers the range, but when several clips draw on the same asset over it the call FAILS and lists them — pass the `clipId` you mean (ids come from getCurrentDocument).", addTrims: @@ -339,6 +343,7 @@ export function buildTools( build("getTranscriptWords", getTranscriptWordsArgs), build("getCursorTrack", getCursorTrackArgs), build("setWordText", setWordTextArgs), + build("removeFillerWords", removeFillerWordsArgs), build("addTrim", addTrimArgs), build("addTrims", addTrimsArgs), build("setTrim", setTrimArgs),