diff --git a/crates/compositor/src/audio.rs b/crates/compositor/src/audio.rs index db9b46e13..f4d0ea819 100644 --- a/crates/compositor/src/audio.rs +++ b/crates/compositor/src/audio.rs @@ -4,7 +4,7 @@ use crate::ffi::*; use crate::regions::SpeedSegment; -use crate::scene::SceneAudio; +use crate::scene::{SceneAudio, SceneAudioTrack}; use anyhow::{bail, Result}; use std::f32::consts::PI; use std::ffi::CString; @@ -914,6 +914,84 @@ pub fn assemble_concatenated_pcm( output } +/// Mix imported audio tracks (issue #350) over the assembled programme. +/// +/// Each track is decoded across its trim window — already resampled to 48 kHz +/// stereo by `decode_clip_audio`, the same path a clip's own audio takes — scaled +/// by its per-track gain (the same `10^(dB/20)` law as `finish_audio`), and summed +/// into the programme at `start_sec`. The programme length is NOT extended: a +/// track that runs past the video is truncated to it, so the audio and video +/// streams stay the same length for the muxer. +/// +/// The decode window is capped up front at the room left in the programme after +/// `start_sec`, and a track starting at/after the end is skipped without decoding. +/// `decode_clip_audio` preallocates from the window, so this keeps a long track +/// pinned near a short programme's end from buffering (and clamping away) hours of +/// PCM. `trim_end_sec` must therefore be concrete — the renderer sends +/// `trimEnd ?? durationSec`. +/// +/// A track whose file has no decodable audio is skipped — the same degradation a +/// stream-less clip gets. +pub fn mix_external_tracks(mut programme: PlanarPcm, tracks: &[SceneAudioTrack]) -> PlanarPcm { + let programme_len = programme.first().map(Vec::len).unwrap_or(0); + if programme_len == 0 { + return programme; + } + for track in tracks { + let offset = (track.start_sec.max(0.0) * AUDIO_OUTPUT_SAMPLE_RATE as f64).round() as usize; + // A track that starts at or past the programme end contributes nothing — + // skip it before decoding anything. + if offset >= programme_len { + continue; + } + let trim_start = track.trim_start_sec.max(0.0); + let Some(trim_end_full) = track.trim_end_sec else { + // Without a concrete end there is no safe window to decode (see the doc + // comment); the renderer always resolves one, so this only guards a + // hand-written scene. + continue; + }; + // Cap the decode window at the room left in the programme. Everything past + // `offset` that overflows is discarded by `overlay_track_pcm` anyway, so + // decoding it only wastes time and memory — a three-hour track placed at + // second 9 of a ten-second export must not buffer three hours of PCM. + let remaining_sec = (programme_len - offset) as f64 / AUDIO_OUTPUT_SAMPLE_RATE as f64; + let trim_end = trim_end_full.min(trim_start + remaining_sec); + if trim_end <= trim_start { + continue; + } + let decoded = match decode_clip_audio(&track.path, trim_start, trim_end) { + Ok(Some(pcm)) => pcm, + _ => continue, + }; + let gain = 10.0f32.powf(track.gain_db.clamp(-12.0, 12.0) / 20.0); + overlay_track_pcm(&mut programme, &decoded, offset, gain); + } + programme +} + +/// Sum one decoded track into the programme at `offset` samples, scaled by `gain`, +/// truncated at the programme's end. Split out of `mix_external_tracks` so the +/// placement/gain/clamp math is testable without ffmpeg, exactly like +/// `mix_aligned_tracks` is split from the decode above. +fn overlay_track_pcm(programme: &mut PlanarPcm, decoded: &PlanarPcm, offset: usize, gain: f32) { + let programme_len = programme.first().map(Vec::len).unwrap_or(0); + if offset >= programme_len { + return; + } + let room = programme_len - offset; + for channel in 0..AUDIO_OUTPUT_CHANNELS { + let Some(source) = decoded.get(channel) else { + continue; + }; + let count = source.len().min(room); + let dst = &mut programme[channel]; + for k in 0..count { + dst[offset + k] += source[k] * gain; + } + } +} + /// Encodeur AAC attaché au muxer avant son header. Les paquets utilisent le même interleaver /// que la vidéo ; les pts restent en unités échantillon jusqu'au rescale vers l'AVStream. pub(crate) struct AacEncoder { @@ -1044,6 +1122,64 @@ mod tests { assert_eq!(mixed[1], vec![0.25, -0.5, 0.75]); } + // Imported audio track overlay (issue #350). + #[test] + fn overlay_sums_at_offset_with_gain() { + let mut programme = planar(&[0.1, 0.1, 0.1, 0.1]); + // ×2 gain, placed at sample offset 1. + overlay_track_pcm(&mut programme, &planar(&[0.2, 0.2]), 1, 2.0); + assert_eq!(programme[0], vec![0.1, 0.5, 0.5, 0.1]); + assert_eq!(programme[1], vec![0.1, 0.5, 0.5, 0.1]); + } + + #[test] + fn overlay_truncates_a_track_that_runs_past_the_programme() { + let mut programme = planar(&[0.0, 0.0, 0.0]); + // A 4-sample track placed at offset 2 has room for only 1 sample. + overlay_track_pcm(&mut programme, &planar(&[1.0, 1.0, 1.0, 1.0]), 2, 1.0); + assert_eq!(programme[0], vec![0.0, 0.0, 1.0]); + } + + #[test] + fn overlay_past_the_end_is_a_no_op() { + let mut programme = planar(&[0.3, 0.3]); + overlay_track_pcm(&mut programme, &planar(&[1.0]), 5, 1.0); + assert_eq!(programme[0], vec![0.3, 0.3]); + } + + #[test] + fn mix_external_tracks_skips_empty_windows() { + let programme = planar(&[0.4, 0.4]); + let tracks = vec![SceneAudioTrack { + path: "/nope.mp3".into(), + start_sec: 0.0, + gain_db: 0.0, + trim_start_sec: 2.0, + trim_end_sec: Some(1.0), // end <= start: empty window, never decoded + }]; + // The empty window is skipped before any decode, so the programme is + // untouched even though the path does not exist. + let out = mix_external_tracks(programme, &tracks); + assert_eq!(out[0], vec![0.4, 0.4]); + } + + #[test] + fn mix_external_tracks_skips_a_track_that_starts_past_the_programme() { + // 2 samples = ~0.00004 s of programme at 48 kHz; the track starts at 1 s, so + // its offset is past the end. It must be skipped before any decode is + // attempted (the path does not exist), never buffering its window. + let programme = planar(&[0.4, 0.4]); + let tracks = vec![SceneAudioTrack { + path: "/nope.mp3".into(), + start_sec: 1.0, + gain_db: 0.0, + trim_start_sec: 0.0, + trim_end_sec: Some(3600.0), + }]; + let out = mix_external_tracks(programme, &tracks); + assert_eq!(out[0], vec![0.4, 0.4]); + } + #[test] fn single_track_is_not_clamped() { // Promesse de non-régression : une source mono-piste ressort telle quelle, y compris diff --git a/crates/compositor/src/pipeline_linux.rs b/crates/compositor/src/pipeline_linux.rs index 910738fc0..8d35c0291 100644 --- a/crates/compositor/src/pipeline_linux.rs +++ b/crates/compositor/src/pipeline_linux.rs @@ -22,7 +22,7 @@ use std::ptr; use crate::audio::{ assemble_concatenated_pcm, build_audio_concat_plan, decode_clip_audio, finish_audio, - stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, + mix_external_tracks, stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, }; use crate::config::Cfg; use crate::d3d::Gpu; @@ -458,6 +458,12 @@ pub fn run_composited_multi( let scene = comp.scene_snapshot(); let audio_settings = scene.as_ref().map(|scene| scene.audio).unwrap_or_default(); + // Imported audio tracks (issue #350), cloned out of the borrowed scene so the + // mix step below owns them. Empty for a project with no imported audio. + let audio_tracks = scene + .as_ref() + .map(|scene| scene.audio_tracks.clone()) + .unwrap_or_default(); // Ring de staging a 2 : l'export ne veut que du debit, une frame de latence // ne se voit pas dans un fichier. Voir `Compositor::set_readback_depth` pour // la raison pour laquelle la preview, elle, reste a 1. @@ -535,7 +541,10 @@ pub fn run_composited_multi( let declared_audio: Vec = clips.iter().map(|c| c.has_audio).collect(); let plan = build_audio_concat_plan(&clip_frame_counts, &declared_audio, out_fps as f64); audio_encoder.encode( - &finish_audio(assemble_concatenated_pcm(&clip_pcm, &plan), audio_settings), + &finish_audio( + mix_external_tracks(assemble_concatenated_pcm(&clip_pcm, &plan), &audio_tracks), + audio_settings, + ), octx, )?; crate::ffi::averr(crate::ffi::av_write_trailer(octx), "write_trailer")?; diff --git a/crates/compositor/src/pipeline_macos.rs b/crates/compositor/src/pipeline_macos.rs index 10de3fac2..1a3ebb9c7 100644 --- a/crates/compositor/src/pipeline_macos.rs +++ b/crates/compositor/src/pipeline_macos.rs @@ -31,7 +31,7 @@ use crate::audio::{ assemble_concatenated_pcm, build_audio_concat_plan, decode_clip_audio, finish_audio, - stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, + mix_external_tracks, stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, }; use crate::compositor::Compositor; use crate::d3d::Gpu; @@ -1077,6 +1077,11 @@ pub fn run_composited_multi( // raconte avoir déjà coûté une fois. let scene = comp.scene_snapshot(); let audio_settings = scene.as_ref().map(|scene| scene.audio).unwrap_or_default(); + // Imported audio tracks (issue #350), cloned out of the borrowed scene. + let audio_tracks = scene + .as_ref() + .map(|scene| scene.audio_tracks.clone()) + .unwrap_or_default(); frames = unsafe { crate::timeline_walk::walk_composited_timeline( clips, @@ -1132,7 +1137,10 @@ pub fn run_composited_multi( let declared_audio: Vec = clips.iter().map(|clip| clip.has_audio).collect(); let plan = build_audio_concat_plan(&clip_frame_counts, &declared_audio, out_fps as f64); audio_encoder.encode( - &finish_audio(assemble_concatenated_pcm(&clip_pcm, &plan), audio_settings), + &finish_audio( + mix_external_tracks(assemble_concatenated_pcm(&clip_pcm, &plan), &audio_tracks), + audio_settings, + ), octx, )?; diff --git a/crates/compositor/src/pipeline_windows.rs b/crates/compositor/src/pipeline_windows.rs index 11738bcd4..01cbda745 100644 --- a/crates/compositor/src/pipeline_windows.rs +++ b/crates/compositor/src/pipeline_windows.rs @@ -4,7 +4,7 @@ use crate::audio::{ assemble_concatenated_pcm, build_audio_concat_plan, decode_clip_audio, finish_audio, - stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, + mix_external_tracks, stretch_clip_pcm_by_speed, AacEncoder, PlanarPcm, }; use crate::compositor::{Compositor, OUT_H, OUT_W}; use crate::config::Cfg; @@ -1343,6 +1343,11 @@ unsafe fn run_multi_inner( // fenêtrage par clip ; `walk_composited_timeline` s'en charge. let scene = comp.scene_snapshot(); let audio_settings = scene.as_ref().map(|scene| scene.audio).unwrap_or_default(); + // Imported audio tracks (issue #350), cloned out of the borrowed scene. + let audio_tracks = scene + .as_ref() + .map(|scene| scene.audio_tracks.clone()) + .unwrap_or_default(); // ---- encodeur (choisi à l'exécution, cf. ExportCodec::candidates) + mux ---- // Backend CPU : pas de pool D3D11 du tout. `av_hwdevice_ctx_init(D3D11VA)` échoue sur @@ -1467,7 +1472,7 @@ unsafe fn run_multi_inner( out_fps as f64, ); let assembled_audio = finish_audio( - assemble_concatenated_pcm(&clip_pcm, &audio_plan), + mix_external_tracks(assemble_concatenated_pcm(&clip_pcm, &audio_plan), &audio_tracks), audio_settings, ); audio_encoder.encode(&assembled_audio, octx)?; diff --git a/crates/compositor/src/scene.rs b/crates/compositor/src/scene.rs index b6fb420c0..e6b8366cf 100644 --- a/crates/compositor/src/scene.rs +++ b/crates/compositor/src/scene.rs @@ -420,6 +420,29 @@ pub struct SceneAudio { pub gain_db: f32, } +/// One imported audio track (issue #350) mixed over the assembled programme — +/// voiceover / BGM / SFX. Deliberately a SEPARATE `Scene` field rather than a +/// member of `SceneAudio`, so `SceneAudio` stays `Copy` and the pipelines keep +/// copying it out of a borrow unchanged. +/// +/// `start_sec` is the track's head on the OUTPUT programme; `trim_start_sec` / +/// `trim_end_sec` window the source file (both source seconds). The renderer +/// resolves `start_sec` from the track's raw timeline position — equal to it when +/// the project has no trims/speed, which is the case this first cut mixes exactly. +#[derive(Debug, Clone, Default, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SceneAudioTrack { + pub path: String, + #[serde(default)] + pub start_sec: f64, + #[serde(default)] + pub gain_db: f32, + #[serde(default)] + pub trim_start_sec: f64, + #[serde(default)] + pub trim_end_sec: Option, +} + #[derive(Debug, Clone, Copy, Deserialize)] #[serde(rename_all = "camelCase")] pub struct SceneOutput { @@ -488,6 +511,10 @@ pub struct Scene { /// Global audio finishing. Default keeps old scene payloads bit-for-bit compatible. #[serde(default)] pub audio: SceneAudio, + /// Imported audio tracks mixed over the programme (issue #350). `#[serde(default)]`: + /// absent from every scene written before this, and from a project with none. + #[serde(default)] + pub audio_tracks: Vec, /// Crop écran par clip, dans le même ordre que `clips` (`cropByClip` côté TS). #[serde(default)] pub crop_by_clip: Vec>, diff --git a/electron/ai-edition/agent-tools.test.ts b/electron/ai-edition/agent-tools.test.ts index 8f8361e59..30e5a34e7 100644 --- a/electron/ai-edition/agent-tools.test.ts +++ b/electron/ai-edition/agent-tools.test.ts @@ -172,6 +172,7 @@ describe("the mutating-tool table", () => { expect([...MUTATING_TOOL_NAMES].sort()).toEqual( [ "addAnnotation", + "addAudio", "addCameraFullscreen", "addSpeed", "addTrim", @@ -184,6 +185,7 @@ describe("the mutating-tool table", () => { "removeTrim", "replaceTimeline", "setAnnotation", + "setAudio", "setCameraFullscreen", "setClipRange", "setSpeed", @@ -2054,3 +2056,167 @@ describe("setZoom answers for the focus it kept", () => { expect(result.resultJson).not.toContain("cursorAnchor"); }); }); + +// Issue #350 — the audio tools. The snapshot and the refusals are what keep the model +// from inventing an asset it cannot import, so both are pinned here beside the +// success paths. +describe("addAudio / setAudio", () => { + /** The fixture plus one imported audio asset. */ + function withAudioAsset(durationSec: number | null = 30): AxcutDocument { + const doc = fixtureDocument(); + return documentSchema.parse({ + ...doc, + assets: [ + ...doc.assets, + { + id: "audio_1", + kind: "audio", + label: "bed.mp3", + originalPath: "C:/audio/bed.mp3", + ...(durationSec == null ? {} : { durationSec }), + }, + ], + }); + } + + it("reports imported audio in the snapshot, with the asset kind beside it", () => { + // Without `kind` the model sees an asset it cannot explain and tries to place it as + // footage; without `audioRanges` it cannot see what is already on the lanes at all. + const placed = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 2, endSec: 6, kind: "voiceover" }), + ); + expect(placed.ok).toBe(true); + const snapshot = executeAgentTool(placed.document as AxcutDocument, "getCurrentDocument", ""); + const parsed = JSON.parse(snapshot.resultJson); + expect(parsed.assets.find((a: { id: string }) => a.id === "audio_1").kind).toBe("audio"); + expect(parsed.audioRanges).toHaveLength(1); + expect(parsed.audioRanges[0]).toMatchObject({ + audioAssetId: "audio_1", + kind: "voiceover", + startSec: 2, + endSec: 6, + }); + }); + + it("anchors the placed region to the clip under it", () => { + const result = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 2, endSec: 6 }), + ); + expect(result.ok).toBe(true); + const region = (result.document as AxcutDocument).audioRanges[0]; + // The anchor is what makes it travel with its clip; a bare startMs/endMs would not. + expect(region.clipId).toBe("clip_1"); + expect(region.sourceStartSec).toBeCloseTo(2, 6); + expect(region.origin).toBe("agent"); + }); + + it("plays the whole file when endSec is omitted", () => { + const result = executeAgentTool( + withAudioAsset(20), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 0, offsetSec: 5 }), + ); + expect(result.ok).toBe(true); + const region = (result.document as AxcutDocument).audioRanges[0]; + // 20s file from an in-point of 5s = 15s of span, so the model never computes it. + expect(region.endMs - region.startMs).toBe(15_000); + }); + + it("refuses an offset at or past the end of a known file", () => { + // Otherwise the omitted-end fallback mints a 0.1s region that plays silence, and the + // model reports it as having placed audio. + const result = executeAgentTool( + withAudioAsset(20), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 0, offsetSec: 20 }), + ); + expect(result.ok).toBe(false); + expect(result.resultJson).toContain("offsetSec"); + }); + + it("allows any offset while the duration is unknown", () => { + // A failed probe leaves no duration; refusing on that would block a legitimate call. + const result = executeAgentTool( + withAudioAsset(null), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 0, offsetSec: 99 }), + ); + expect(result.ok).toBe(true); + }); + + it("refuses an unknown asset and names the audio the project actually has", () => { + const result = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "nope", startSec: 0, endSec: 4 }), + ); + expect(result.ok).toBe(false); + expect(result.resultJson).toContain("audio_1"); + }); + + it("refuses a video asset, pointing at the tool that does place footage", () => { + const result = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "asset_1", startSec: 0, endSec: 4 }), + ); + expect(result.ok).toBe(false); + expect(result.resultJson).toContain("replaceTimeline"); + }); + + it("setAudio re-levels and re-lanes the pill it names", () => { + const placed = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 2, endSec: 6 }), + ); + const id = JSON.parse(placed.resultJson).audioId; + const result = executeAgentTool( + placed.document as AxcutDocument, + "setAudio", + JSON.stringify({ audioId: id, gainDb: -6, kind: "voiceover" }), + ); + expect(result.ok).toBe(true); + expect((result.document as AxcutDocument).audioRanges[0]).toMatchObject({ + gainDb: -6, + kind: "voiceover", + }); + }); + + it("setAudio applies the same offset guard as addAudio", () => { + const placed = executeAgentTool( + withAudioAsset(20), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 2, endSec: 6 }), + ); + const id = JSON.parse(placed.resultJson).audioId; + const result = executeAgentTool( + placed.document as AxcutDocument, + "setAudio", + JSON.stringify({ audioId: id, offsetSec: 25 }), + ); + expect(result.ok).toBe(false); + }); + + it("removeModifier deletes an audio region by id, like every other kind", () => { + const placed = executeAgentTool( + withAudioAsset(), + "addAudio", + JSON.stringify({ audioAssetId: "audio_1", startSec: 2, endSec: 6 }), + ); + const id = JSON.parse(placed.resultJson).audioId; + const result = executeAgentTool( + placed.document as AxcutDocument, + "removeModifier", + JSON.stringify({ id }), + ); + expect(result.ok).toBe(true); + expect((result.document as AxcutDocument).audioRanges).toEqual([]); + // The asset went with the last region that played it. + expect((result.document as AxcutDocument).assets.some((a) => a.id === "audio_1")).toBe(false); + }); +}); diff --git a/electron/ai-edition/agent-tools.ts b/electron/ai-edition/agent-tools.ts index 5a0966398..fa774b680 100644 --- a/electron/ai-edition/agent-tools.ts +++ b/electron/ai-edition/agent-tools.ts @@ -337,6 +337,32 @@ function droppedByEdit(before: AxcutDocument, after: AxcutDocument) { // private — callers only ever need the composed `*Args`.) const secondsSchema = z.number().finite().nonnegative(); +/** Span given to an agent-placed audio region when the asset has no probed duration yet. + * Short on purpose: a wrong guess the user has to shorten beats one that silently covers + * the whole programme. */ +const DEFAULT_AGENT_AUDIO_SEC = 10; + +/** + * "Start the file at `offsetSec`" is only answerable when there IS file left at that + * point. Past the end it yields a region that plays silence, which the model then reports + * as having placed audio — the failure worth a refusal rather than a shrug. + * + * A non-positive `durationSec` is UNKNOWN, not zero: an import whose probe failed carries + * 0 until the renderer re-probes it, and refusing on that would block a legitimate call. + */ +function audioOffsetRefusal( + asset: { id: string; durationSec?: number | null } | undefined, + offsetSec: number, +): string | null { + const duration = asset?.durationSec; + if (duration == null || !(duration > 0)) return null; + if (offsetSec < duration) return null; + return ( + `offsetSec ${offsetSec}s is at or past the end of ${asset?.id} (${duration}s), ` + + "so the region would play nothing. Pick an offset inside the file." + ); +} + export const addTrimArgs = z.object({ startSec: secondsSchema, endSec: secondsSchema, @@ -479,6 +505,24 @@ export const setAnnotationArgs = z.object({ text: z.string().optional(), }); +export const addAudioArgs = z.object({ + audioAssetId: z.string().min(1), + startSec: secondsSchema, + endSec: secondsSchema.optional(), + kind: z.enum(["voiceover", "music"]).default("music"), + offsetSec: secondsSchema.default(0), + gainDb: z.number().min(-60).max(12).default(0), +}); + +export const setAudioArgs = z.object({ + audioId: z.string().min(1), + startSec: secondsSchema.optional(), + endSec: secondsSchema.optional(), + kind: z.enum(["voiceover", "music"]).optional(), + offsetSec: secondsSchema.optional(), + gainDb: z.number().min(-60).max(12).optional(), +}); + export const addCameraFullscreenArgs = z.object({ startSec: secondsSchema, endSec: secondsSchema, @@ -541,6 +585,8 @@ export const OPENSCREEN_TOOL_NAMES = [ "setAnnotation", "addCameraFullscreen", "setCameraFullscreen", + "addAudio", + "setAudio", "removeTrim", "removeModifier", "removeClip", @@ -607,6 +653,8 @@ export const MUTATING_TOOL_NAMES: ReadonlySet = new Set([ "setAnnotation", "addCameraFullscreen", "setCameraFullscreen", + "addAudio", + "setAudio", "removeTrim", "removeModifier", "removeClip", @@ -665,7 +713,9 @@ export function documentSnapshotForModel( const autoFocusAll = legacy?.autoFocusAll === true; return { timeBaseNote: - "clips and trims are in source-time seconds; zooms, speedRegions, annotations and cameraFullscreenRegions are in virtual (edited-timeline) seconds.", + "clips and trims are in source-time seconds; zooms, speedRegions, annotations, cameraFullscreenRegions and audioRanges are in virtual (edited-timeline) seconds.", + audioNote: + "audioRanges are imported voiceover / music files laid over the recording. They are regions like every other kind — anchored to the clip they cover, so they travel with it through reorder and trim — and they play at 1x whatever a speed region does to the picture under them. addAudio places an EXISTING asset of kind 'audio'; nothing here can import a file from disk, so if the project has no audio asset, say so rather than inventing an id.", zoomNote: `renderedScale is what the viewer sees (depth is an ordinal, not a factor: ${ZOOM_DEPTH_LEGEND}). ` + "When a zoom carries customScale it wins over depth and depthIsOverridden is true — " + @@ -683,6 +733,10 @@ export function documentSnapshotForModel( assets: document.assets.map((a) => ({ id: a.id, label: a.label, + // "audio" is an imported voiceover / music file: it is never a clip, it is played + // by an audio region. Without this the model sees an asset it cannot explain and + // tries to place it on the timeline as footage. + kind: a.kind, durationSec: a.durationSec ?? null, hasCameraTrack: a.cameraTrack != null, cameraVisible: a.cameraTrack?.visible ?? false, @@ -755,6 +809,21 @@ export function documentSnapshotForModel( startSec: roundSec(c.startMs), endSec: roundSec(c.endMs), })), + // Imported audio, coalesced to whole pills like every other kind so the model + // reasons about what the user sees on the ruler rather than about the fragments a + // clip boundary happens to have split it into. + audioRanges: coalesceForAgent(document.audioRanges).map((a) => ({ + id: a.id, + startSec: roundSec(a.startMs), + endSec: roundSec(a.endMs), + audioAssetId: a.audioAssetId, + // Which lane it sits on, and part of its identity: changing it moves the region + // between lanes rather than creating a second one. + kind: a.kind, + // Where in the FILE the region starts playing, in that file's own seconds. + offsetSec: a.offsetSec, + gainDb: a.gainDb, + })), hasTranscript: document.transcripts.length > 0 || document.transcript !== null, }; } @@ -1843,6 +1912,127 @@ export function executeAgentTool( }; } + case "addAudio": { + const parsed = addAudioArgs.safeParse(args); + if (!parsed.success) return failure(parsed.error.message); + const { audioAssetId, kind, offsetSec, gainDb } = parsed.data; + const asset = document.assets.find((a) => a.id === audioAssetId); + // Two distinct refusals, because they need two different corrections: an unknown + // id is a hallucinated asset, a video id is the model reaching for footage. + if (!asset) { + const available = document.assets.filter((a) => a.kind === "audio"); + return failure( + `Unknown asset: ${audioAssetId}.` + + (available.length + ? ` Imported audio in this project: ${available.map((a) => `${a.id} (${a.label})`).join(", ")}.` + : " This project has no imported audio; a file can only be imported from the editor, not from here."), + ); + } + if (asset.kind !== "audio") { + return failure( + `Asset ${audioAssetId} is video, not audio. addAudio plays an imported audio file over the recording; to place footage use replaceTimeline.`, + ); + } + const offsetRefusal = audioOffsetRefusal(asset, offsetSec); + if (offsetRefusal) return failure(offsetRefusal); + // No endSec means "as long as the file is" — the natural span, and the one the + // editor's own add uses. Falling back to the asset duration here rather than + // making the model compute it keeps the two paths on one rule. + const startSec = parsed.data.startSec; + const endSec = + parsed.data.endSec ?? + startSec + Math.max(0.1, (asset.durationSec ?? DEFAULT_AGENT_AUDIO_SEC) - offsetSec); + const startMs = toMs(Math.min(startSec, endSec)); + const endMs = toMs(Math.max(startSec, endSec)); + const region = { + id: createId("audio"), + startMs, + endMs, + audioAssetId, + kind, + offsetSec, + gainDb, + origin: "agent" as const, + }; + const placed = anchorForAgent(region, document, "audio"); + const landing = landingOf(placed, document); + if (!landing.anchored) { + return coversNoClip("audio", startMs / 1000, endMs / 1000, document); + } + const next: AxcutDocument = { + ...document, + audioRanges: [...document.audioRanges, ...placed] as AxcutDocument["audioRanges"], + }; + return { + ok: true, + document: next, + resultJson: JSON.stringify({ + audioId: landing.ids[0], + ...landingReport(landing, startMs / 1000, endMs / 1000), + }), + summary: + `added ${kind} "${asset.label}" ${formatSec(landing.startSec)} – ${formatSec(landing.endSec)}` + + landingSuffix(landing, startMs / 1000, endMs / 1000), + }; + } + + case "setAudio": { + const parsed = setAudioArgs.safeParse(args); + if (!parsed.success) return failure(parsed.error.message); + const { audioId } = parsed.data; + const existing = document.audioRanges.find((a) => a.id === audioId); + if (!existing) return failure(`Unknown audio region: ${audioId}`); + if (parsed.data.offsetSec !== undefined) { + const refusal = audioOffsetRefusal( + document.assets.find((a) => a.id === existing.audioAssetId), + parsed.data.offsetSec, + ); + if (refusal) return failure(refusal); + } + const audioPill = new Set(resolvePillIds(document.audioRanges, audioId)); + const { startMs, endMs } = resolveSpanMs(existing, parsed.data.startSec, parsed.data.endSec); + // The payload patch hits EVERY fragment under the pill before the span is + // replaced: kind, offset and gain are all part of the region identity, so + // patching one fragment of a ventilated region would split it into two pills. + const patched = document.audioRanges.map((a) => + audioPill.has(a.id) + ? { + ...a, + ...(parsed.data.kind !== undefined ? { kind: parsed.data.kind } : {}), + ...(parsed.data.offsetSec !== undefined ? { offsetSec: parsed.data.offsetSec } : {}), + ...(parsed.data.gainDb !== undefined ? { gainDb: parsed.data.gainDb } : {}), + } + : a, + ); + const rebuilt = replacePillSpan( + patched, + audioId, + startMs, + endMs, + document.timeline.clips, + () => createId("audio"), + ); + const landing = landingAfterPillEdit(document.audioRanges, rebuilt, audioPill, document); + if (!landing.anchored) { + return coversNoClip("audio", startMs / 1000, endMs / 1000, document); + } + const next: AxcutDocument = { + ...document, + audioRanges: rebuilt as AxcutDocument["audioRanges"], + }; + return { + ok: true, + document: next, + resultJson: JSON.stringify({ + audioId: landing.ids[0], + ...landingReport(landing, startMs / 1000, endMs / 1000), + }), + summary: + `updated audio ${audioId} ${formatSec(landing.startSec)} – ${formatSec(landing.endSec)}` + + landingSuffix(landing, startMs / 1000, endMs / 1000), + }; + } + case "removeTrim": { const parsed = removeTrimArgs.safeParse(args); if (!parsed.success) return failure(parsed.error.message); @@ -1872,9 +2062,10 @@ export function executeAgentTool( else if (document.annotations.some((a) => a.id === id)) kind = "annotation"; else if (speedRegions.some((s) => s.id === id)) kind = "speed"; else if (cameraFullscreenRegions.some((c) => c.id === id)) kind = "cameraFullscreen"; + else if (document.audioRanges.some((a) => a.id === id)) kind = "audio"; if (!kind) { return failure( - `No zoom / speed / annotation / full-camera modifier with id ${id}. ` + + `No zoom / speed / annotation / full-camera / audio modifier with id ${id}. ` + `For a trim use removeTrim; for a clip use removeClip.`, ); } diff --git a/electron/ai-edition/deep-agent/service.test.ts b/electron/ai-edition/deep-agent/service.test.ts index 7729bd624..2e5f66167 100644 --- a/electron/ai-edition/deep-agent/service.test.ts +++ b/electron/ai-edition/deep-agent/service.test.ts @@ -73,6 +73,11 @@ const ARGS: Record = { setAnnotation: { annotationId: "ann_nope" }, addCameraFullscreen: { startSec: 1, endSec: 2 }, setCameraFullscreen: { cameraFullscreenId: "cam_nope" }, + // The fixture has no `kind: "audio"` asset, so this exercises the refusal branch — + // which is the honest one to pin: the agent cannot import a file, only place one the + // project already has. + addAudio: { audioAssetId: "audio_nope", startSec: 1, endSec: 2 }, + setAudio: { audioId: "audio_nope" }, removeTrim: { trimRangeId: "trim_1" }, removeModifier: { id: "nope" }, removeClip: { clipId: "clip_1" }, diff --git a/electron/ai-edition/deep-agent/service.ts b/electron/ai-edition/deep-agent/service.ts index d804a3b1a..c862eab70 100644 --- a/electron/ai-edition/deep-agent/service.ts +++ b/electron/ai-edition/deep-agent/service.ts @@ -27,6 +27,7 @@ import type { AxcutDocument } from "../../../src/lib/ai-edition/schema"; import { ZOOM_DEPTH_LEGEND } from "../../../src/lib/ai-edition/timeline/zoom-scale"; import { addAnnotationArgs, + addAudioArgs, addCameraFullscreenArgs, addSpeedArgs, addTrimArgs, @@ -45,6 +46,7 @@ import { replaceTimelineArgs, resolveCursorAssetId, setAnnotationArgs, + setAudioArgs, setCameraFullscreenArgs, setClipRangeArgs, setSpeedArgs, @@ -115,6 +117,7 @@ const BASE_SYSTEM_PROMPT = [ "- Silences, pauses and dead stretches are removed as trims INSIDE the placed clip. Send them together with addTrims once you know the ranges; addTrim is for a single cut or a correction. The placed clip stays the canonical cut; it is not rebuilt to drop them.", "- Changing where a clip starts or ends within its source is setClipRange — the clip's in/out, distinct from a trim.", `- addZoom takes a virtual-timeline span (depth is an ordinal 1–6 selecting from a fixed table — ${ZOOM_DEPTH_LEGEND} — never a multiplier; focus in 0–1 frame fractions). addSpeed changes pacing over a span. addAnnotation puts text on screen. addCameraFullscreen enlarges the webcam, and only does something where assets[].hasCameraTrack is true.`, + "- addAudio lays an imported voiceover or music file over a span. It plays an asset the project already has (kind 'audio'); importing a file from disk is the editor's job, not a tool you have — so when the project has none, say so rather than naming an id that does not exist.", "- moveClip changes the order of placed clips, one call per clip that moves, preserving ids, source ranges, trims and anchored effects. replaceTimeline rebuilds the timeline from kept intervals and sorts them, so it cannot reorder anything.", "- Deleting is a first-class action, not a workaround: removeTrim, removeModifier, removeClip. Never fake a deletion by re-adding an element or zeroing it out (span 0, speed 1×) — that leaves it in the document and misreports what you did.", "If nothing in the list does what was asked, say so; do not approximate it with a bigger tool.", @@ -170,10 +173,14 @@ export const TOOL_DESCRIPTIONS: Record = { "Add a camera-fullscreen region over a span of the edited timeline (virtual seconds): the webcam fills the frame for that span. This only does something when the footage under that span comes from an asset with a linked webcam — check assets[].hasCameraTrack (or hasAnyCamera) in getCurrentDocument first. On footage with no camera the call is refused rather than storing a region that would render nothing; say so instead of retrying.", setCameraFullscreen: "Move or resize an existing camera-fullscreen region by id (virtual-timeline seconds). Only the fields you pass are changed. Refused if the new span lands on footage with no linked webcam.", + addAudio: + "Lay an ALREADY-IMPORTED audio file over the recording across a span of the edited timeline (virtual seconds): a voiceover, or a music bed. audioAssetId must name an asset whose kind is 'audio' — getCurrentDocument lists them; nothing here can import a file from disk, so if there is none, say so instead of guessing an id. Omit endSec to play the whole file from offsetSec. kind picks the lane ('voiceover' or 'music'): two regions on the SAME lane may not overlap, two on different lanes may, which is how a voiceover sits over a bed. offsetSec is where in the FILE playback starts, gainDb its level (0 is unchanged, negative ducks it).", + setAudio: + "Move, resize, re-level, re-lane, or re-point an existing audio region by id (virtual-timeline seconds). Only the fields you pass are changed. Use it to duck a bed under narration (gainDb), to shift what part of the file plays (offsetSec), or to move it between the voiceover and music lanes (kind).", removeTrim: "Delete a trim range by id — the cut is undone and that span plays/exports again. This is how you 'remove a trim'; never re-add a trim to undo one.", removeModifier: - "Delete a modifier (zoom / speed / annotation / camera-fullscreen) by id; the kind is resolved from the id. This is how you 'remove'/'delete' one — never neutralise it (span 0, speed 1×), which leaves it in the document. For a trim use removeTrim; for a clip use removeClip.", + "Delete a modifier (zoom / speed / annotation / camera-fullscreen / audio) by id; the kind is resolved from the id. This is how you 'remove'/'delete' one — never neutralise it (span 0, speed 1×), which leaves it in the document. For a trim use removeTrim; for a clip use removeClip.", removeClip: "Delete a placed clip by id; remaining clips close the gap and effects anchored to it are dropped. Use only when the user asks to remove a clip — to shorten one, use setClipRange.", }; @@ -339,6 +346,8 @@ export function buildTools( build("setAnnotation", setAnnotationArgs), build("addCameraFullscreen", addCameraFullscreenArgs), build("setCameraFullscreen", setCameraFullscreenArgs), + build("addAudio", addAudioArgs), + build("setAudio", setAudioArgs), build("removeTrim", removeTrimArgs), build("removeModifier", removeModifierArgs), build("removeClip", removeClipArgs), diff --git a/electron/ai-edition/document-service.test.ts b/electron/ai-edition/document-service.test.ts index 6cbdb97c9..d4d842325 100644 --- a/electron/ai-edition/document-service.test.ts +++ b/electron/ai-edition/document-service.test.ts @@ -280,6 +280,48 @@ describe("DocumentService", () => { expect(after.project.primaryAssetId).toBe(first.project.primaryAssetId); expect(after.assets).toHaveLength(2); }); + + // Issue #350 — external audio import (voiceover / BGM / SFX). + it("appends an audio asset without claiming the primary slot", async () => { + const doc = await service.createProject("P"); + const updated = await service.addAsset(doc.project.id, { + path: "/tmp/voiceover.mp3", + kind: "audio", + }); + expect(updated.assets).toHaveLength(1); + expect(updated.assets[0]?.kind).toBe("audio"); + // An audio-only file must never become the project's primary asset, even + // when it is the first file added to an otherwise-empty project. + expect(updated.project.primaryAssetId).toBeUndefined(); + }); + + it("keeps the existing video primary when an audio track is added", async () => { + const doc = await service.createProject("P"); + const withVideo = await service.addAsset(doc.project.id, { path: "/tmp/screen.mp4" }); + const primary = withVideo.project.primaryAssetId; + const withAudio = await service.addAsset(doc.project.id, { + path: "/tmp/bgm.wav", + kind: "audio", + }); + expect(withAudio.project.primaryAssetId).toBe(primary); + expect(withAudio.assets).toHaveLength(2); + }); + + it("rejects unsupported audio extensions", async () => { + const doc = await service.createProject("P"); + await expect( + service.addAsset(doc.project.id, { path: "/tmp/clip.mp4", kind: "audio" }), + ).rejects.toBeInstanceOf(ProjectFileError); + }); + + it("accepts a video extension under the default kind but not as audio", async () => { + const doc = await service.createProject("P"); + // The same extension routing works in reverse: an .mp3 is fine as audio + // but rejected as video (covered above), and an .mp4 is the opposite. + await expect( + service.addAsset(doc.project.id, { path: "/tmp/a.mp3", kind: "audio" }), + ).resolves.toBeDefined(); + }); }); describe("removeAsset", () => { @@ -363,6 +405,51 @@ describe("DocumentService", () => { expect(after.project.primaryAssetId).toBe(b.assets[1]?.id); }); + // Issue #350 — an audio overlay can never be primary. + it("passes primary to the next VIDEO asset, never to an audio asset", async () => { + const doc = await service.createProject("P"); + const video = await service.addAsset(doc.project.id, { path: "/tmp/screen.mp4" }); + await service.addAsset(doc.project.id, { path: "/tmp/music.mp3", kind: "audio" }); + const primaryId = video.project.primaryAssetId; + expect(primaryId).toBeTruthy(); + // Removing the only video leaves just the audio asset; primary must clear, + // not fall to the audio one. + const after = await service.removeAsset(doc.project.id, primaryId ?? ""); + expect(after.project.primaryAssetId).toBeUndefined(); + expect(after.assets).toHaveLength(1); + expect(after.assets[0]?.kind).toBe("audio"); + }); + + it("drops audio regions that played a removed audio asset", async () => { + const doc = await service.createProject("P"); + await service.addAsset(doc.project.id, { path: "/tmp/screen.mp4" }); + const withAudio = await service.addAsset(doc.project.id, { + path: "/tmp/music.mp3", + kind: "audio", + }); + const audioId = withAudio.assets.find((a) => a.kind === "audio")?.id ?? ""; + expect(audioId).toBeTruthy(); + const withTrack = await service.saveProject({ + ...withAudio, + audioRanges: [ + { + id: "audio_1", + startMs: 0, + endMs: 10_000, + audioAssetId: audioId, + kind: "music", + offsetSec: 0, + gainDb: 0, + origin: "user", + }, + ], + }); + expect(withTrack.audioRanges).toHaveLength(1); + const after = await service.removeAsset(doc.project.id, audioId); + expect(after.audioRanges).toEqual([]); + expect(after.assets.some((a) => a.id === audioId)).toBe(false); + }); + it("resequences other assets and rederives their anchored regions", async () => { const created = await service.createProject("P"); const withA = await service.addAsset(created.project.id, { path: "/tmp/a.mp4" }); diff --git a/electron/ai-edition/document-service.ts b/electron/ai-edition/document-service.ts index 3c93e3bc0..19280930a 100644 --- a/electron/ai-edition/document-service.ts +++ b/electron/ai-edition/document-service.ts @@ -38,6 +38,9 @@ export interface ProjectSummary { export interface AddAssetInput { path: string; label?: string; + // "audio" imports an external voiceover / BGM / SFX file (issue #350). + // Defaults to "video" when omitted, so existing callers are unaffected. + kind?: "video" | "audio"; } export class DocumentNotFoundError extends Error { @@ -72,6 +75,24 @@ function isSupportedVideoPath(filePath: string): boolean { return SUPPORTED_VIDEO_EXTENSIONS.has(ext); } +// Imported audio (issue #350). Decoding is handled downstream by the same +// WebCodecs / ffmpeg paths that read a video's audio track, so this list is the +// container formats decodeAudioData and the compositor can open. +const SUPPORTED_AUDIO_EXTENSIONS = new Set([ + ".mp3", + ".wav", + ".m4a", + ".aac", + ".flac", + ".ogg", + ".opus", +]); + +function isSupportedAudioPath(filePath: string): boolean { + const ext = path.extname(filePath).toLowerCase(); + return SUPPORTED_AUDIO_EXTENSIONS.has(ext); +} + function safeProjectId(raw: string): string { // ponytail: project ids are uuid-prefixed strings (e.g. "proj_"). Reject // anything that smells like path traversal before we ever touch the disk. @@ -283,7 +304,15 @@ export class DocumentService { if (!input.path) { throw new ProjectFileError("Asset path is required.", projectId); } - if (!isSupportedVideoPath(input.path)) { + const kind = input.kind ?? "video"; + if (kind === "audio") { + if (!isSupportedAudioPath(input.path)) { + throw new ProjectFileError( + `Unsupported audio extension: ${path.extname(input.path)} (supported: ${[...SUPPORTED_AUDIO_EXTENSIONS].join(", ")})`, + projectId, + ); + } + } else if (!isSupportedVideoPath(input.path)) { throw new ProjectFileError( `Unsupported video extension: ${path.extname(input.path)} (supported: ${[...SUPPORTED_VIDEO_EXTENSIONS].join(", ")})`, projectId, @@ -300,18 +329,24 @@ export class DocumentService { } const asset: AxcutAsset = { id: createId("asset"), - kind: "video", + kind, label: input.label?.trim() || path.basename(absolutePath), originalPath: absolutePath, sizeBytes, cameraTrack: null, }; + // An audio import is an overlay, never the thing the timeline is built + // around, so it must not claim the empty primaryAssetId slot — otherwise the + // first file dropped into a fresh project (a BGM track) would become its + // primary asset and the editor would try to lay out clips from a file with + // no video. + const claimsPrimary = kind !== "audio" && !doc.project.primaryAssetId; const next: AxcutDocument = { ...doc, assets: [...doc.assets, asset], project: { ...doc.project, - ...(doc.project.primaryAssetId ? {} : { primaryAssetId: asset.id }), + ...(claimsPrimary ? { primaryAssetId: asset.id } : {}), updatedAt: new Date().toISOString(), }, }; @@ -324,9 +359,13 @@ export class DocumentService { throw new ProjectFileError(`Asset ${assetId} not found in project ${projectId}.`, projectId); } const assets = doc.assets.filter((a) => a.id !== assetId); + // Primary is the thing the timeline is built around, so it must fall to the + // next VIDEO asset — never an audio overlay (issue #350), which can't be + // primary (see addAsset). Falling back to `assets[0]` would hand primary to + // an audio asset when the removed one was the last video. const primaryAssetId = doc.project.primaryAssetId === assetId - ? (assets[0]?.id ?? undefined) + ? (assets.find((a) => a.kind !== "audio")?.id ?? undefined) : doc.project.primaryAssetId; const withoutAssetClips = doc.timeline.clips .filter((clip) => clip.assetId === assetId) @@ -334,6 +373,9 @@ export class DocumentService { const next: AxcutDocument = { ...withoutAssetClips, assets, + // Drop audio regions that played the removed asset — they would otherwise + // dangle, pointing at an asset the document no longer has. + audioRanges: withoutAssetClips.audioRanges.filter((r) => r.audioAssetId !== assetId), timeline: { ...withoutAssetClips.timeline, trimRanges: withoutAssetClips.timeline.trimRanges.filter((r) => r.assetId !== assetId), diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index e140a4e37..f6132b4f6 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -289,6 +289,14 @@ interface Window { name?: string; canceled?: boolean; }>; + // Import an external audio file from the timeline toolbar (issue #350). + openAudioFilePicker: () => Promise<{ + success: boolean; + path?: string; + name?: string; + canceled?: boolean; + message?: string; + }>; setCurrentVideoPath: (path: string) => Promise<{ success: boolean }>; setCurrentRecordingSession: ( session: import("../src/lib/recordingSession").RecordingSession | null, diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index aa2014670..b3ad77fab 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -186,6 +186,32 @@ function hasAllowedImportVideoExtension(filePath: string): boolean { return ALLOWED_IMPORT_VIDEO_EXTENSIONS.has(path.extname(filePath).toLowerCase()); } +// Imported audio (issue #350). Kept separate from the video set so the two +// pickers stay honest — an audio picker must not approve a video path and vice +// versa. Mirrors SUPPORTED_AUDIO_EXTENSIONS in the document service. +const ALLOWED_IMPORT_AUDIO_EXTENSIONS = new Set([ + ".mp3", + ".wav", + ".m4a", + ".aac", + ".flac", + ".ogg", + ".opus", +]); + +function hasAllowedImportAudioExtension(filePath: string): boolean { + return ALLOWED_IMPORT_AUDIO_EXTENSIONS.has(path.extname(filePath).toLowerCase()); +} + +// Video OR audio. The type-specific pickers stay honest (see the audio set's +// comment), but the generic media READS — peaks, binary, file-info, chunk — serve +// whichever kind the document points at, so they must accept both. Gating them on +// video alone dropped every imported audio path once `approvedPaths` was empty +// (a project reopen), and the waveform was lost for good (issue #350). +function hasAllowedImportMediaExtension(filePath: string): boolean { + return hasAllowedImportVideoExtension(filePath) || hasAllowedImportAudioExtension(filePath); +} + function runProcess( command: string, args: string[], @@ -282,8 +308,13 @@ async function prepareSupplementalPreviewAudioTrack(videoPath: string) { return { success: true, path: pathToFileURL(outputPath).toString() }; } -async function approveReadableVideoPath( - filePath?: string | null, +// Shared core behind the media path approvers. `hasAllowedExtension` is the ONLY +// thing that differs between video and audio imports, so it is the single knob: +// an already-approved path passes regardless, otherwise the extension gate, +// optional trusted-dir confinement, and a stat check decide whether to approve. +async function approveReadableMediaPath( + filePath: string | null | undefined, + hasAllowedExtension: (p: string) => boolean, trustedDirs?: string[], ): Promise { const normalizedPath = normalizeVideoSourcePath(filePath); @@ -295,7 +326,7 @@ async function approveReadableVideoPath( return normalizedPath; } - if (!hasAllowedImportVideoExtension(normalizedPath)) { + if (!hasAllowedExtension(normalizedPath)) { return null; } @@ -322,6 +353,29 @@ async function approveReadableVideoPath( return normalizedPath; } +function approveReadableVideoPath( + filePath?: string | null, + trustedDirs?: string[], +): Promise { + return approveReadableMediaPath(filePath, hasAllowedImportVideoExtension, trustedDirs); +} + +function approveReadableAudioPath( + filePath?: string | null, + trustedDirs?: string[], +): Promise { + return approveReadableMediaPath(filePath, hasAllowedImportAudioExtension, trustedDirs); +} + +// For the generic media reads that accept either kind — NOT for the pickers, +// which must stay type-specific (see `hasAllowedImportMediaExtension`). +function approveReadableAvPath( + filePath?: string | null, + trustedDirs?: string[], +): Promise { + return approveReadableMediaPath(filePath, hasAllowedImportMediaExtension, trustedDirs); +} + function resolveRecordingOutputPath(fileName: string): string { const trimmed = fileName.trim(); if (!trimmed) { @@ -3590,6 +3644,8 @@ export function registerIpcHandlers( } }); + // The media tab imports VIDEO (it arranges clips). Audio is imported from the + // timeline toolbar instead (issue #350) — see `open-audio-file-picker` below. ipcMain.handle("open-video-file-picker", async () => { try { const dialogOptions = buildDialogOptions( @@ -3636,6 +3692,55 @@ export function registerIpcHandlers( } }); + // Import an external audio file (voiceover / BGM / SFX) — issue #350. Driven by + // the timeline's "Add audio" tool: audio is a timeline overlay (like an + // annotation), not a media-tab clip, so it has its own audio-only picker and the + // renderer adds it as a kind:"audio" asset + track at the playhead. + ipcMain.handle("open-audio-file-picker", async () => { + try { + const dialogOptions = buildDialogOptions( + { + title: mainT("dialogs", "fileDialogs.selectAudio"), + defaultPath: RECORDINGS_DIR, + filters: [ + { + name: mainT("dialogs", "fileDialogs.audioFiles"), + extensions: ["mp3", "wav", "m4a", "aac", "flac", "ogg", "opus"], + }, + { name: mainT("dialogs", "fileDialogs.allFiles"), extensions: ["*"] }, + ], + properties: ["openFile"], + }, + getMainWindow(), + ); + const result = await dialog.showOpenDialog(dialogOptions); + + if (result.canceled || result.filePaths.length === 0) { + return { success: false, canceled: true }; + } + + const normalizedPath = await approveReadableAudioPath(result.filePaths[0]); + if (!normalizedPath) { + return { + success: false, + message: "Selected file is not a supported readable audio file", + }; + } + + return { + success: true, + path: normalizedPath, + }; + } catch (error) { + console.error("Failed to open audio file picker:", error); + return { + success: false, + message: "Failed to open audio file picker", + error: String(error), + }; + } + }); + ipcMain.handle("reveal-in-folder", async (_, filePath: string) => { try { // showItemInFolder returns nothing, it throws on error @@ -3661,7 +3766,7 @@ export function registerIpcHandlers( ipcMain.handle("read-binary-file", async (_, filePath: string) => { try { - const normalizedPath = await approveReadableVideoPath(filePath); + const normalizedPath = await approveReadableAvPath(filePath); if (!normalizedPath) { return { success: false, @@ -3691,7 +3796,7 @@ export function registerIpcHandlers( // recording above that can never be loaded whole — see read-file-chunk). ipcMain.handle("get-readable-file-info", async (_, filePath: string) => { try { - const normalizedPath = await approveReadableVideoPath(filePath); + const normalizedPath = await approveReadableAvPath(filePath); if (!normalizedPath) { return { success: false, @@ -3727,7 +3832,7 @@ export function registerIpcHandlers( async (_, filePath: string, durationSec: number): Promise => { try { // Same approval gate as every other read of a renderer-supplied path. - const normalizedPath = await approveReadableVideoPath(filePath); + const normalizedPath = await approveReadableAvPath(filePath); if (!normalizedPath) { return { success: false, message: "File path is not approved" }; } @@ -3751,7 +3856,7 @@ export function registerIpcHandlers( // do (2 GiB cap) and a 16 GB machine cannot hold for multi-GB recordings. ipcMain.handle("read-file-chunk", async (_, filePath: string, offset: number, length: number) => { try { - const normalizedPath = await approveReadableVideoPath(filePath); + const normalizedPath = await approveReadableAvPath(filePath); if (!normalizedPath) { return { success: false, diff --git a/electron/ipc/nativeBridge.ts b/electron/ipc/nativeBridge.ts index 5d4992c10..7b01e0b2f 100644 --- a/electron/ipc/nativeBridge.ts +++ b/electron/ipc/nativeBridge.ts @@ -489,6 +489,7 @@ export function registerNativeBridgeHandlers(context: NativeBridgeContext) { request.payload.projectId, request.payload.path, request.payload.label, + request.payload.kind, ), ); case "document.removeAsset": diff --git a/electron/native-bridge/services/aiEditionService.ts b/electron/native-bridge/services/aiEditionService.ts index 0fbbccc9c..90088781a 100644 --- a/electron/native-bridge/services/aiEditionService.ts +++ b/electron/native-bridge/services/aiEditionService.ts @@ -151,9 +151,17 @@ export class AiEditionService { } } - async addAsset(projectId: string, path: string, label?: string): Promise { - const document = await this.options.documents.addAsset(projectId, { path, label }); - const assetId = document.project.primaryAssetId ?? document.assets.at(-1)?.id ?? ""; + async addAsset( + projectId: string, + path: string, + label?: string, + kind?: "video" | "audio", + ): Promise { + const document = await this.options.documents.addAsset(projectId, { path, label, kind }); + // The just-added asset is always the last one; primaryAssetId is only a + // fallback for the video case and would point at the wrong asset for an + // audio import (which never claims primary), so prefer the tail. + const assetId = document.assets.at(-1)?.id ?? document.project.primaryAssetId ?? ""; return { assetId, document }; } diff --git a/electron/preload.ts b/electron/preload.ts index 6aff16407..16227f110 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -276,6 +276,9 @@ contextBridge.exposeInMainWorld("electronAPI", { openVideoFilePicker: () => { return ipcRenderer.invoke("open-video-file-picker"); }, + openAudioFilePicker: () => { + return ipcRenderer.invoke("open-audio-file-picker"); + }, setCurrentVideoPath: (path: string) => { return ipcRenderer.invoke("set-current-video-path", path); }, diff --git a/src/components/ai-edition/EditorEmptyState.test.tsx b/src/components/ai-edition/EditorEmptyState.test.tsx index acb513e4b..e175bd6bc 100644 --- a/src/components/ai-edition/EditorEmptyState.test.tsx +++ b/src/components/ai-edition/EditorEmptyState.test.tsx @@ -53,6 +53,7 @@ const sampleDoc = vi.hoisted( }, annotations: [], zoomRanges: [], + audioRanges: [], legacyEditor: null, }), ); diff --git a/src/components/ai-edition/ExportDialog.showInFolder.test.tsx b/src/components/ai-edition/ExportDialog.showInFolder.test.tsx index 835d9c8fd..f08d03362 100644 --- a/src/components/ai-edition/ExportDialog.showInFolder.test.tsx +++ b/src/components/ai-edition/ExportDialog.showInFolder.test.tsx @@ -75,6 +75,7 @@ const DOC: AxcutDocument = { }, annotations: [], zoomRanges: [], + audioRanges: [], legacyEditor: null, }; diff --git a/src/components/ai-edition/ExportDialog.test.ts b/src/components/ai-edition/ExportDialog.test.ts index 3aa8d85b5..41557e7ec 100644 --- a/src/components/ai-edition/ExportDialog.test.ts +++ b/src/components/ai-edition/ExportDialog.test.ts @@ -57,6 +57,7 @@ function doc(assets: AxcutAsset[], clips: AxcutClip[]): AxcutDocument { }, annotations: [], zoomRanges: [], + audioRanges: [], legacyEditor: null, }; } diff --git a/src/components/ai-edition/NewEditorShell.tsx b/src/components/ai-edition/NewEditorShell.tsx index b371d100c..7977581b4 100644 --- a/src/components/ai-edition/NewEditorShell.tsx +++ b/src/components/ai-edition/NewEditorShell.tsx @@ -327,7 +327,12 @@ export function NewEditorShell() { }; }, [promptUnsaved, saveDocument]); - const videoSources = useMemo(() => { + // Every asset with a resolvable URL, split below. It is deliberately NOT handed to + // anything as-is: `videoSources` reaching a consumer that treats its entries as + // footage is how an imported mp3 became a timeline clip (an audio-only project has no + // `primaryAssetId`, and both `handleLoadedMetadata` and `replaceTimeline` fall back to + // `assets[0]`, which the preview had already mounted as a