Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
121 changes: 121 additions & 0 deletions electron/ai-edition/agent-tools.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -183,6 +183,7 @@ describe("the mutating-tool table", () => {
"removeClip",
"removeModifier",
"removeTrim",
"removeFillerWords",
"replaceTimeline",
"setAnnotation",
"setAudio",
Expand Down Expand Up @@ -2331,6 +2332,126 @@ describe("getTranscriptWords", () => {
});
});

describe("removeFillerWords", () => {
function repeatedWordDocument(): AxcutDocument {
const base = fixtureDocument();
return {
...base,
transcripts: [
{
...base.transcripts[0],
words: [
{ id: "filler", segmentId: "seg_1", startSec: 1, endSec: 1.2, text: "like" },
{ id: "meaningful", segmentId: "seg_1", startSec: 3, endSec: 3.2, text: "like" },
],
},
],
};
}

it("cuts only the selected occurrence and reports the trim actually stored", () => {
const before = repeatedWordDocument();
const read = run(before, "getTranscriptWords", {});
expect(JSON.parse(read.resultJson).words.map((word: { id: string }) => word.id)).toEqual([
"filler",
"meaningful",
]);
const result = run(before, "removeFillerWords", { wordIds: ["filler"] });
expect(result.ok).toBe(true);
const next = result.document as AxcutDocument;
const trim = next.timeline.trimRanges.at(-1);
expect(trim).toMatchObject({ assetId: "asset_1", clipId: "clip_1", origin: "agent" });
expect(trim?.endSec).toBeLessThan(3); // the second "like" still plays
expect(next.transcripts).toEqual(before.transcripts);
expect(next.timeline.clips).toEqual(before.timeline.clips);
const payload = JSON.parse(result.resultJson);
expect(payload).toMatchObject({
requested: 1,
removedCount: 1,
removed: [
{
wordId: "filler",
text: "like",
assetId: "asset_1",
clipId: "clip_1",
wordStartSec: 1,
wordEndSec: 1.2,
trimRangeId: trim?.id,
startSec: trim?.startSec,
endSec: trim?.endSec,
},
],
});
expect(next.timeline.trimRanges).toHaveLength(before.timeline.trimRanges.length + 1);
});

it("refuses an invalid ID atomically, including when another ID is valid", () => {
const before = repeatedWordDocument();
const result = run(before, "removeFillerWords", { wordIds: ["filler", "missing"] });
expect(result.ok).toBe(false);
expect(result.document).toBeUndefined();
expect(result.resultJson).toContain("Unknown transcript word ID: missing");
expect(before.timeline.trimRanges).toHaveLength(1);
});

it("resolves a word ID to its own asset and clip, not the primary asset", () => {
const before = repeatedWordDocument();
before.assets.push({ ...before.assets[0], id: "asset_2" });
before.timeline.clips.push({
...before.timeline.clips[0],
id: "clip_3",
assetId: "asset_2",
});
before.transcripts.push({
...before.transcripts[0],
assetId: "asset_2",
words: [{ id: "other_filler", segmentId: "seg_1", startSec: 4, endSec: 4.2, text: "um" }],
});
const result = run(before, "removeFillerWords", { wordIds: ["other_filler"] });
expect(result.ok).toBe(true);
expect(result.document?.timeline.trimRanges.at(-1)).toMatchObject({
assetId: "asset_2",
clipId: "clip_3",
});
expect(JSON.parse(result.resultJson).removed[0]).toMatchObject({
assetId: "asset_2",
clipId: "clip_3",
});
});

it("refuses invalid timestamps without guessing a span", () => {
const before = repeatedWordDocument();
before.transcripts[0].words[0].endSec = Number.NaN;
const result = run(before, "removeFillerWords", { wordIds: ["filler"] });
expect(result.ok).toBe(false);
expect(result.document).toBeUndefined();
expect(result.resultJson).toContain("invalid source timestamps");
});

it("refuses a word covered by two clips", () => {
const before = repeatedWordDocument();
before.timeline.clips.push({
...before.timeline.clips[0],
id: "duplicate_clip",
});
const result = run(before, "removeFillerWords", { wordIds: ["filler"] });
expect(result.ok).toBe(false);
expect(result.document).toBeUndefined();
expect(result.resultJson).toContain("exactly one clip; found 2");
});

it("honours project edit consent", () => {
const result = executeAgentTool(
repeatedWordDocument(),
"removeFillerWords",
JSON.stringify({ wordIds: ["filler"] }),
{ editsAllowed: false },
);
expect(result.document).toBeUndefined();
expect(JSON.parse(result.resultJson).code).toBe("consent_required");
});
});

describe("setWordText", () => {
it("changes the text and leaves the timeline alone", () => {
const before = documentWithWords();
Expand Down
152 changes: 152 additions & 0 deletions electron/ai-edition/agent-tools.ts
Original file line number Diff line number Diff line change
Expand Up @@ -414,6 +414,10 @@ export const addTrimsArgs = z.object({
ranges: z.array(z.union([addTrimArgs, z.unknown()])).min(1),
});

export const removeFillerWordsArgs = z.object({
wordIds: z.array(z.string().min(1)).min(1),
});

export const setTrimArgs = z.object({
trimRangeId: z.string().min(1),
startSec: secondsSchema,
Expand Down Expand Up @@ -599,6 +603,7 @@ export const OPENSCREEN_TOOL_NAMES = [
"getTranscriptWords",
"getCursorTrack",
"setWordText",
"removeFillerWords",
"addTrim",
"addTrims",
"setTrim",
Expand Down Expand Up @@ -670,6 +675,7 @@ export const MUTATING_TOOL_NAMES: ReadonlySet<string> = new Set([
// Writes the transcript, not the timeline — but it writes the document, so it is a
// consented edit like any other.
"setWordText",
"removeFillerWords",
"addTrim",
"addTrims",
"addZooms",
Expand Down Expand Up @@ -1406,6 +1412,152 @@ export function executeAgentTool(
};
}

case "removeFillerWords": {
const parsed = removeFillerWordsArgs.safeParse(args);
if (!parsed.success) return failure(parsed.error.message);
const { wordIds } = parsed.data;
if (new Set(wordIds).size !== wordIds.length) {
return failure("Duplicate word IDs in one request. Nothing was modified.");
}

// Resolve every ID before writing anything. Word IDs, rather than guessed
// timestamps, designate which occurrence of a repeated word is speech to cut.
const transcripts = [...document.transcripts];
if (
document.transcript &&
!transcripts.some((transcript) => transcript.assetId === document.transcript?.assetId)
) {
transcripts.push(document.transcript);
}
const selected: Array<{
wordId: string;
text: string;
assetId: string;
clipId: string;
startSec: number;
endSec: number;
}> = [];
for (const wordId of wordIds) {
const matches = transcripts.flatMap((transcript) =>
transcript.words
.filter((word) => word.id === wordId)
.map((word) => ({ assetId: transcript.assetId, word })),
);
if (matches.length !== 1) {
return failure(
matches.length === 0
? `Unknown transcript word ID: ${wordId}. Call getTranscriptWords to read the IDs. Nothing was modified.`
: `Word ID ${wordId} appears more than once in the transcripts. Nothing was modified.`,
);
}
const { assetId, word } = matches[0];
const asset = document.assets.find((candidate) => candidate.id === assetId);
if (!asset || isGeneratedAssetId(assetId) || word.source === "synth") {
return failure(`Word ${wordId} has no recorded source asset. Nothing was modified.`);
}
const { startSec, endSec } = word;
if (
!Number.isFinite(startSec) ||
!Number.isFinite(endSec) ||
startSec < 0 ||
endSec <= startSec ||
(asset.durationSec !== undefined &&
(!Number.isFinite(asset.durationSec) || endSec > asset.durationSec))
) {
return failure(`Word ${wordId} has invalid source timestamps. Nothing was modified.`);
}
const clips = document.timeline.clips.filter((clip) => {
const clipEnd = clip.sourceEndSec ?? asset.durationSec;
return (
clip.assetId === assetId &&
clipEnd !== undefined &&
Number.isFinite(clipEnd) &&
startSec >= clip.sourceStartSec &&
endSec <= clipEnd
);
});
if (clips.length !== 1) {
return failure(
`Word ${wordId} must be fully contained in exactly one clip; found ${clips.length}. Nothing was modified.`,
);
}
const clip = clips[0];
if (
document.timeline.trimRanges.some(
(trim) =>
trimAppliesToClip(trim, clip) && startSec < trim.endSec && endSec > trim.startSec,
)
) {
return failure(`Word ${wordId} is already cut by a trim. Nothing was modified.`);
}
selected.push({ wordId, text: word.text, assetId, clipId: clip.id, startSec, endSec });
}

let current = document;
const removed: Array<Record<string, unknown>> = [];
for (const word of selected) {
const execution = executeAgentTool(
current,
"addTrim",
JSON.stringify({
assetId: word.assetId,
clipId: word.clipId,
startSec: word.startSec,
endSec: word.endSec,
reason: "filler word",
}),
options,
);
if (!execution.ok || !execution.document) {
return failure(
`Could not cut word ${word.wordId}: ${execution.resultJson}. Nothing was modified.`,
);
}
const { trimRangeId } = JSON.parse(execution.resultJson) as { trimRangeId: string };
const trim = execution.document.timeline.trimRanges.find(
(range) => range.id === trimRangeId,
);
const clip = current.timeline.clips.find((candidate) => candidate.id === word.clipId);
const asset = current.assets.find((candidate) => candidate.id === word.assetId);
const clipEnd = clip?.sourceEndSec ?? asset?.durationSec;
if (
!trim ||
!clip ||
clipEnd === undefined ||
trim.clipId !== word.clipId ||
!Number.isFinite(trim.startSec) ||
!Number.isFinite(trim.endSec) ||
trim.startSec < clip.sourceStartSec ||
trim.endSec > clipEnd ||
trim.endSec <= trim.startSec
) {
return failure(`Cut for word ${word.wordId} left its clip. Nothing was modified.`);
}
current = execution.document;
removed.push({
wordId: word.wordId,
text: word.text,
assetId: word.assetId,
clipId: word.clipId,
wordStartSec: word.startSec,
wordEndSec: word.endSec,
trimRangeId,
startSec: trim.startSec,
endSec: trim.endSec,
});
}
return {
ok: true,
document: current,
resultJson: JSON.stringify({
requested: wordIds.length,
removedCount: removed.length,
removed,
}),
summary: `removed ${removed.length} filler word${removed.length === 1 ? "" : "s"}`,
};
}

case "addTrim": {
const parsed = addTrimArgs.safeParse(args);
if (!parsed.success) return failure(parsed.error.message);
Expand Down
1 change: 1 addition & 0 deletions electron/ai-edition/deep-agent/service.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,7 @@ const ARGS: Record<string, unknown> = {
getTranscriptWords: {},
getCursorTrack: {},
setWordText: { wordId: "word_1", text: "Hullo" },
removeFillerWords: { wordIds: ["word_1"] },
addTrim: { startSec: 1, endSec: 2 },
addTrims: { ranges: [{ startSec: 1, endSec: 2 }] },
setTrim: { trimRangeId: "trim_1", startSec: 1, endSec: 2 },
Expand Down
9 changes: 7 additions & 2 deletions electron/ai-edition/deep-agent/service.ts
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@ import {
isMutatingTool,
moveClipArgs,
removeClipArgs,
removeFillerWordsArgs,
removeModifierArgs,
removeTrimArgs,
replaceTimelineArgs,
Expand Down Expand Up @@ -93,7 +94,7 @@ const CONSENT_PROMPT_BLOCK = [
"",
"PROJECT EDITS ARE CURRENTLY DISABLED by the user, who asked to be consulted before the timeline changes.",
"- Read freely: getCurrentDocument and getTranscript work as usual.",
"- Do NOT call any tool that writes (addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.",
"- Do NOT call any tool that writes (removeFillerWords, addTrim, setTrim, setClipRange, moveClip, replaceTimeline, add*/set* effects, remove*). Every one of them will be refused, so calling them wastes the turn and tells the user nothing.",
"- Instead: say precisely what you would change — which tool, which times, which ids — and ask the user to confirm. Be specific enough that they can say yes to it.",
"- Never state or imply that an edit was applied. If the user confirms and you are still refused, tell them the 'Project edits' setting in Settings → AI has to be re-enabled first.",
].join("\n");
Expand All @@ -117,6 +118,7 @@ const BASE_SYSTEM_PROMPT = [
// other than English. Say what the tool does; let the model do the matching.
"How the tools map to intent — pick the most specific one, and prefer the smallest edit that satisfies the request:",
"- Silences, pauses and dead stretches are removed as trims INSIDE the placed clip. Send them together with addTrims once you know the ranges; addTrim is for a single cut or a correction. The placed clip stays the canonical cut; it is not rebuilt to drop them.",
"- Only when the user explicitly asks to remove filler words: read getTranscriptWords, decide which occurrences are fillers from their surrounding speech, then pass ONLY those word IDs to removeFillerWords. The tool resolves their exact source and clip, makes trims, and reports the actual cuts. Do not remove every occurrence of a word just because its text matches a filler elsewhere.",
"- Changing where a clip starts or ends within its source is setClipRange — the clip's in/out, distinct from a trim.",
`- addZoom takes a virtual-timeline span (depth is an ordinal 1–6 selecting from a fixed table — ${ZOOM_DEPTH_LEGEND} — never a multiplier; focus in 0–1 frame fractions). addSpeed changes pacing over a span. addAnnotation puts text on screen. addCameraFullscreen enlarges the webcam, and only does something where assets[].hasCameraTrack is true.`,
"- addAudio lays an imported voiceover or music file over a span. It plays an asset the project already has (kind 'audio'); importing or recording one is the editor's job, not a tool you have — so when the project has none, say so rather than naming an id that does not exist.",
Expand Down Expand Up @@ -149,9 +151,11 @@ export const TOOL_DESCRIPTIONS: Record<string, string> = {
getCursorTrack:
"Read the recorded pointer track for an asset: where the cursor was over time, downsampled to a readable rate. Each point carries atSec (the asset's own source clock), virtualSec (the same instant on the edited timeline — the coordinate addZoom takes, null when no clip carries it), cx/cy as 0–1 fractions of the frame, and `shape`, an index into the pointer bitmaps the recording used (equal values are the same pointer; a change means the pointer changed, e.g. arrow to text caret). Points that are not plain moves carry `kind`; points a trim cuts out of playback carry `trimmed`. These are real samples, not a summary — reading what the pointer was doing is yours. Omit assetId for the primary asset. It answers `available:false` in two DIFFERENT ways you must not confuse: reason 'no-sidecar' means this asset was checked and genuinely has no telemetry, while reason 'unavailable' means it could not be read from here.",
getTranscriptWords:
'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText takes; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.',
'Read the transcript one WORD at a time for an asset: each word\'s id, text, start/end seconds, and — only when it is not plain transcription — `source` ("user" for a word the user corrected, "synth" for one they typed in) and `originalText` (what the transcriber had heard before the correction). This is the ONLY read that gives you the ids setWordText and removeFillerWords take; getTranscript answers in segments, whose ids belong to a different namespace and are not accepted there. A whole transcript is large, so pass startSec/endSec to read just the passage you mean to fix. Omit assetId for the primary asset.',
setWordText:
"Correct ONE word's text, by the id getTranscriptWords returns. This changes the TRANSCRIPT and nothing else: the captions follow it, the film is untouched and no audio is cut. Use it when the transcriber misheard something — a name, a technical term — and the user asks for it to read correctly. Passing an empty string BLANKS the word: it keeps its place in the media but leaves the captions, which is how a junk token like \"(inaudible)\" is removed without cutting the speech around it. Writing the transcriber's own text back clears the correction. This is NOT how you make a spoken word go away — that removes only the label and leaves the film saying it; use addTrim, which cuts the audio with it.",
removeFillerWords:
"Remove only the spoken filler-word occurrences selected by `wordIds` from getTranscriptWords. Use ONLY when the user explicitly asks for filler-word cleanup. Read words in context and choose IDs yourself; the executor does not classify speech or remove every matching text. It resolves each ID to one recorded asset, valid source timestamps, and exactly one fully covering clip, then adds non-destructive trims. If any ID or target is invalid or ambiguous, nothing changes. The result lists each word and the ACTUAL trim span (which addTrim may shape), asset, clip and trim ID; report only those results.",
addTrim:
"Add ONE trim range: a cut of a span inside a clip (this source-time span will not be played or exported) that does NOT split the clip. Times are in seconds of the asset's source time. This is the preferred (and for 'remove silences' requests, the only) way to handle silences; it preserves the user's placed clips and only adds a cut. When you have several cuts to make, use addTrims and send them together — this one is for a single cut or a later correction. A cut belongs to ONE clip: `clipId` is inferred when a single clip covers the range, but when several clips draw on the same asset over it the call FAILS and lists them — pass the `clipId` you mean (ids come from getCurrentDocument).",
addTrims:
Expand Down Expand Up @@ -339,6 +343,7 @@ export function buildTools(
build("getTranscriptWords", getTranscriptWordsArgs),
build("getCursorTrack", getCursorTrackArgs),
build("setWordText", setWordTextArgs),
build("removeFillerWords", removeFillerWordsArgs),
build("addTrim", addTrimArgs),
build("addTrims", addTrimsArgs),
build("setTrim", setTrimArgs),
Expand Down
Loading