Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions packages/engine/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -213,6 +213,7 @@ export {
extractionFrameCountForDuration,
resolveProjectRelativeSrc,
getFrameAtTime,
isVideoHiddenBeforeStreamStart,
createFrameLookupTable,
FrameLookupTable,
analyzeClipMediaFit,
Expand Down
9 changes: 1 addition & 8 deletions packages/engine/src/services/audioFxRender.ts
Original file line number Diff line number Diff line change
Expand Up @@ -176,14 +176,7 @@ export function writeWav(
writeFileSync(path, buf);
}

/**
* Split an interleaved buffer into one array per channel.
*
* The graph used to fold everything to mono, which collapsed a stereo bed's
* width for the render only — and cost ~3 dB through the very mono-to-stereo
* rematrix that `prepareAudioTrack`'s pan filter exists to avoid. Preview kept
* the track stereo, so the two diverged the moment any effect was enabled.
*/
/** Split an interleaved buffer into one array per channel, keeping stereo width as preview does. */
function deinterleave(samples: Float32Array, channels: number): Float32Array[] {
if (channels <= 1) return [samples];
const frames = Math.floor(samples.length / channels);
Expand Down
180 changes: 49 additions & 131 deletions packages/engine/src/services/audioMixer.ts
Original file line number Diff line number Diff line change
@@ -1,15 +1,9 @@
// fallow-ignore-file complexity code-duplication
/**
* Audio Mixer Service
*
* Processes and mixes audio tracks using FFmpeg.
*/

import { isSelfOrAncestorHidden, memberGroupKey, isMemberGroupHidden } from "./mediaHidden.js";
import { closeSync, existsSync, mkdirSync, mkdtempSync, openSync, rmSync, writeFileSync } from "fs";
import { join, dirname, isAbsolute, relative } from "path";
import { parseHTML } from "linkedom";
import { extractAudioMetadata } from "../utils/ffprobe.js";
import { extractAudioMetadata, type AudioMetadata } from "../utils/ffprobe.js";
import { isNotMediaPayload } from "../utils/notMediaPayload.js";
import { clampAudioGain } from "@hyperframes/core/audio-gain";
import { clampFadesToDuration, readElementFades } from "@hyperframes/core/audio-fade";
Expand Down Expand Up @@ -61,19 +55,8 @@ import type { AudioVolumeKeyframe } from "./audioMixer.types.js";

export type { AudioElement, MixResult } from "./audioMixer.types.js";

/**
* Filename every caller must use for the mixed-audio artifact.
*
* The extension is load-bearing, not cosmetic: FFmpeg picks the muxer from it,
* and the mix is AAC-encoded. A raw ADTS `.aac` stream has nowhere to record
* the encoder's priming delay, so those leading samples decode as real silence
* and shift the whole track ~1024 samples (21.33 ms at 48 kHz) late against a
* frame-accurate video track. An MP4-family container carries the delay as an
* edit list, which every decoder then strips, so the mix lands on its authored
* start. Keep the choice here rather than at each call site: the same file is
* muxed into the video, shipped in a distributed plan, and handed to users as
* the PNG-sequence sidecar, and all three have to agree.
*/
/** An MP4-family container records AAC priming as an edit list; a raw .aac plays ~21 ms late.
* The video mux, the distributed plan and the PNG-sequence sidecar all read this one name. */
export const MIXED_AUDIO_FILENAME = "audio.m4a";

function clampVolume(volume: number): number {
Expand Down Expand Up @@ -129,12 +112,13 @@ const MAX_RAMP_SLICES = 240;
function buildRampFilterComplex(
lane: RateSpec & object,
duration: number,
headFilter: string | null,
tailFilter: string | null,
): string {
const slices = Math.min(MAX_RAMP_SLICES, Math.max(1, Math.ceil(duration / RAMP_SLICE_SECONDS)));
const step = duration / slices;
const split = Array.from({ length: slices }, (_, i) => `[s${i}]`).join("");
const parts = [`[0:a]asplit=${slices}${split}`];
const parts = [`[0:a]${headFilter ? `${headFilter},` : ""}asplit=${slices}${split}`];
for (let i = 0; i < slices; i += 1) {
const from = sourceTimeAt(lane, i * step);
const to = sourceTimeAt(lane, (i + 1) * step);
Expand All @@ -156,27 +140,20 @@ function buildRampFilterComplex(
}

function preparedAudioOutputArgs(
srcPath: string,
metadata: AudioMetadata | null,
playbackRate: RateSpec,
duration = 0,
): Promise<string[]> {
return stereoOutputArgs(srcPath).then((channelArgs) => {
const filters: string[] = [];
const outputArgs: string[] = [];
if (channelArgs[0] === "-af" && channelArgs[1]) {
filters.push(channelArgs[1]);
} else {
outputArgs.push(...channelArgs);
}
if (typeof playbackRate === "object") {
const graph = buildRampFilterComplex(playbackRate, duration, filters.join(",") || null);
return ["-filter_complex", graph, "-map", "[out]", ...outputArgs];
}
const atempo = buildAtempoFilter(playbackRate);
if (atempo) filters.push(atempo);
if (filters.length > 0) outputArgs.push("-af", filters.join(","));
return outputArgs;
});
duration: number,
headFilter: string | null,
): string[] {
const channelFilter = metadata?.channels === 1 ? STEREO_CHANNEL_FILTER : null;
const outputArgs = channelFilter ? [] : ["-ac", "2"];
if (typeof playbackRate === "object") {
const graph = buildRampFilterComplex(playbackRate, duration, headFilter, channelFilter);
return ["-filter_complex", graph, "-map", "[out]", ...outputArgs];
}
const filters = [headFilter, channelFilter, buildAtempoFilter(playbackRate)].filter(Boolean);
if (filters.length > 0) outputArgs.push("-af", filters.join(","));
return outputArgs;
}

function escapeExpressionCommas(expression: string): string {
Expand Down Expand Up @@ -217,16 +194,6 @@ const VOLUME_SIMPLIFY_EPSILON = 0.005;
// native stereo sources have FL/FR and pass through unchanged.
const STEREO_CHANNEL_FILTER = "pan=stereo|FL=FL+FC|FR=FR+FC";

async function stereoOutputArgs(srcPath: string): Promise<string[]> {
try {
const { channels } = await extractAudioMetadata(srcPath);
if (channels === 1) return ["-af", STEREO_CHANNEL_FILTER];
} catch {
// Preserve the previous FFmpeg conversion path when metadata probing fails.
}
return ["-ac", "2"];
}

/**
* Reduce a sorted keyframe list to a perceptually-equivalent piecewise-linear
* envelope with a bounded segment count.
Expand Down Expand Up @@ -654,65 +621,12 @@ export function parseAudioElements(html: string): AudioElement[] {
return elements;
}

async function extractAudioFromVideo(
videoPath: string,
outputPath: string,
options?: { startTime?: number; duration?: number; playbackRate?: RateSpec },
signal?: AbortSignal,
config?: Partial<Pick<EngineConfig, "ffmpegProcessTimeout">>,
): Promise<ExtractResult> {
const ffmpegProcessTimeout = config?.ffmpegProcessTimeout ?? DEFAULT_CONFIG.ffmpegProcessTimeout;
const outputDir = dirname(outputPath);
if (!existsSync(outputDir)) mkdirSync(outputDir, { recursive: true });

const playbackRate = normalizeRateSpec(options?.playbackRate);
const args: string[] = [];
if (options?.startTime !== undefined) args.push("-ss", String(options.startTime));
if (options?.duration !== undefined) {
args.push("-t", String(sourceTimeAt(playbackRate, options.duration)));
}
args.push("-i", videoPath);
const outputArgs = await preparedAudioOutputArgs(videoPath, playbackRate, options?.duration);
args.push("-vn", "-acodec", "pcm_s16le", "-ar", "48000", ...outputArgs);
if (playbackRate !== 1 && options?.duration !== undefined) {
args.push("-t", String(options.duration));
}
args.push("-y", outputPath);

const result = await runFfmpeg(args, { signal, timeout: ffmpegProcessTimeout });

if (signal?.aborted) {
const failure: AudioProcessingFailure = {
stage: "cancelled",
reason: "cancelled",
owner: "user",
retryable: false,
detail: "Audio extract cancelled",
};
return {
success: false,
outputPath,
durationMs: result.durationMs,
error: failure.detail,
failure,
};
}
if (!result.success) {
const failure = ffmpegFailure("extract", result);
return {
success: false,
outputPath,
durationMs: result.durationMs,
error: failure.detail,
failure,
};
}
return { success: true, outputPath, durationMs: result.durationMs };
}

async function prepareAudioTrack(
/** Decode `duration` s of audio from `mediaStart` in Chrome's media time. Until every stream has
* started, an input seek snaps audio to the video's first keyframe, so read by timestamp instead. */
async function extractAudioSegment(
srcPath: string,
outputPath: string,
stage: "extract" | "prepare",
mediaStart: number,
duration: number,
playbackRate: RateSpec = 1,
Expand All @@ -723,21 +637,25 @@ async function prepareAudioTrack(
const outputDir = dirname(outputPath);
if (!existsSync(outputDir)) mkdirSync(outputDir, { recursive: true });
const normalizedPlaybackRate = normalizeRateSpec(playbackRate);
const outputArgs = await preparedAudioOutputArgs(srcPath, normalizedPlaybackRate, duration);
const sourceDuration = sourceTimeAt(normalizedPlaybackRate, duration);
// A failed probe keeps FFmpeg's default channel conversion and the input seek.
const metadata = await extractAudioMetadata(srcPath).catch(() => null);
const inputSeek = mediaStart >= (metadata?.latestStreamLeadSeconds ?? 0);
const start = formatFilterNumber(mediaStart);
const timestampTrim = inputSeek
? null
: `atrim=start=${start}:end=${formatFilterNumber(mediaStart + sourceDuration)},asetpts=PTS-${start}/TB,aresample=async=1:first_pts=0`;
const outputArgs = preparedAudioOutputArgs(
metadata,
normalizedPlaybackRate,
duration,
timestampTrim,
);

const args = [
"-ss",
String(mediaStart),
"-t",
String(sourceTimeAt(normalizedPlaybackRate, duration)),
"-i",
srcPath,
"-acodec",
"pcm_s16le",
"-ar",
"48000",
...outputArgs,
];
const args = inputSeek
? ["-ss", String(mediaStart), "-t", String(sourceDuration), "-i", srcPath]
: ["-i", srcPath];
args.push("-vn", "-acodec", "pcm_s16le", "-ar", "48000", ...outputArgs);
if (normalizedPlaybackRate !== 1) args.push("-t", String(duration));
args.push("-y", outputPath);

Expand All @@ -749,7 +667,7 @@ async function prepareAudioTrack(
reason: "cancelled",
owner: "user",
retryable: false,
detail: "Audio prepare cancelled",
detail: `Audio ${stage} cancelled`,
};
return {
success: false,
Expand All @@ -759,7 +677,7 @@ async function prepareAudioTrack(
failure,
};
}
const failure = !result.success ? ffmpegFailure("prepare", result) : undefined;
const failure = !result.success ? ffmpegFailure(stage, result) : undefined;
return {
success: result.success,
outputPath,
Expand Down Expand Up @@ -1243,14 +1161,13 @@ export async function processCompositionAudio(
let audioSrcPath = srcPath;
if (element.type === "video") {
const extractedPath = join(workDir, `${element.id}-extracted.wav`);
const extractResult = await extractAudioFromVideo(
const extractResult = await extractAudioSegment(
srcPath,
extractedPath,
{
startTime: element.mediaStart,
duration: element.end - element.start,
playbackRate: element.playbackRate,
},
"extract",
element.mediaStart,
element.end - element.start,
element.playbackRate,
effectiveSignal,
config,
);
Expand All @@ -1272,9 +1189,10 @@ export async function processCompositionAudio(
audioSrcPath = extractedPath;
} else {
const trimmedPath = join(workDir, `${element.id}-trimmed.wav`);
const prepResult = await prepareAudioTrack(
const prepResult = await extractAudioSegment(
srcPath,
trimmedPath,
"prepare",
element.mediaStart,
element.end - element.start,
element.playbackRate,
Expand Down
2 changes: 1 addition & 1 deletion packages/engine/src/services/audioVolumeEnvelope.ts
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@
* output (and the golden baselines) only change where a fade is actually applied.
*
* The prepared tracks are always `pcm_s16le`, 48 kHz, stereo (see
* `prepareAudioTrack` / `extractAudioFromVideo`). Anything else is rejected so
* `extractAudioSegment`). Anything else is rejected so
* the caller can fall back to the expression path rather than corrupting audio.
*/

Expand Down
Loading
Loading