mirror of
https://github.com/ZHANGTIANYAO1/teamspeak-music-bot.git
synced 2026-10-02 04:52:50 +08:00
fix(local): remux aac into .m4a so the extracted audio is bit-exact (#149)
Extraction always used Matroska (.mka) because it takes essentially any audio codec. That is right for most codecs but wrong for AAC: MP4 records the AAC encoder priming (the ~1000 warm-up samples every AAC encoder emits) in an edit list, and the edit list does not survive into Matroska. The remuxed track then decodes ~23 ms longer than the source, with the priming samples played at the head instead of discarded. Measured on a 5s 640x480 fixture: source audio decodes to 962980 bytes of PCM, the .mka to 967440 — 4460 bytes / ~23 ms extra, peaking at -66 dBFS. Inaudible in practice, but it also puts the track fractionally out of step with its own reported duration, for no reason. Pick the container by codec instead: aac -> .m4a (keeps the edit list), everything else -> .mka as before. If the preferred container refuses the codec, retry into .mka before falling back to keeping the whole video. AAC is worth the special case because mp4 / mov / m4v — what people actually upload — almost always carry it. Adds the strongest available test of the "lossless" claim: decode the audio straight out of the source mp4, decode the stored extract, assert the PCM is byte-for-byte equal. Forcing .mka fails it with exactly the 4460-byte delta. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
1 parent
9fdc164f98
commit
af1dac848d
6 files changed
+855
-10
No files matched your search
@@ -84,6 +84,7 @@ describe("LocalMusicProvider: source video cannot be deleted after extraction (#
|
||||
// Fell back to the original container — the extract was discarded.
|
||||
expect(resolved!.url.endsWith(".mp4")).toBe(true);
|
||||
expect(existsSync(resolved!.url)).toBe(true);
|
||||
expect(existsSync(resolved!.url.replace(/\.mp4$/, ".m4a"))).toBe(false);
|
||||
expect(existsSync(resolved!.url.replace(/\.mp4$/, ".mka"))).toBe(false);
|
||||
|
||||
const onDisk = statSync(resolved!.url).size;
|
||||
|
||||
+33
-1
@@ -392,7 +392,9 @@ describe("LocalMusicProvider video upload, end to end (#149)", () => {
|
||||
const resolved = await p.getSongUrl(song.id);
|
||||
expect(resolved).not.toBeNull();
|
||||
// The video container is gone; what remains is the extracted audio track.
|
||||
expect(resolved!.url.endsWith(".mka")).toBe(true);
|
||||
// AAC (what libx264+aac mp4s carry) goes to .m4a so the encoder-priming
|
||||
// edit list survives — see extractedAudioExt.
|
||||
expect(resolved!.url.endsWith(".m4a")).toBe(true);
|
||||
expect(existsSync(join(dir, `${song.id}.mp4`))).toBe(false);
|
||||
expect(existsSync(resolved!.url)).toBe(true);
|
||||
expect(statSync(resolved!.url).size).toBeGreaterThan(0);
|
||||
@@ -418,6 +420,36 @@ describe("LocalMusicProvider video upload, end to end (#149)", () => {
|
||||
expect(decoded.stdout.length).toBeGreaterThan(300000);
|
||||
}, 60000);
|
||||
|
||||
it.runIf(have)("aac extraction decodes bit-for-bit identically to the audio inside the video", async () => {
|
||||
// The strongest statement of "lossless": decode the audio track straight
|
||||
// out of the source mp4, decode the stored extract, compare the PCM.
|
||||
// A Matroska remux would NOT pass this — it loses the MP4 edit list that
|
||||
// discards AAC encoder priming, so it decodes ~23 ms longer.
|
||||
const p = new LocalMusicProvider(dir);
|
||||
const bytes = render("bitexact.mp4", withAudio(4, "libx264", "aac"));
|
||||
const sourceCopy = join(dir, "source-kept.mp4");
|
||||
writeFileSync(sourceCopy, bytes);
|
||||
|
||||
const song = await p.uploadAudio({
|
||||
buffer: bytes, originalName: "bitexact.mp4", mimeType: "video/mp4",
|
||||
});
|
||||
const url = (await p.getSongUrl(song.id))!.url;
|
||||
|
||||
const toPcm = (input: string, pre: string[] = []) => spawnSync(
|
||||
ffmpeg!,
|
||||
["-hide_banner", "-loglevel", "error", "-i", input, ...pre,
|
||||
"-f", "s16le", "-ar", "48000", "-ac", "2", "-acodec", "pcm_s16le", "-"],
|
||||
{ maxBuffer: 128 * 1024 * 1024 },
|
||||
);
|
||||
|
||||
const fromVideo = toPcm(sourceCopy, ["-vn", "-map", "0:a:0"]);
|
||||
const fromExtract = toPcm(url);
|
||||
expect(fromVideo.status).toBe(0);
|
||||
expect(fromExtract.status).toBe(0);
|
||||
expect(fromExtract.stdout.length).toBe(fromVideo.stdout.length);
|
||||
expect(fromExtract.stdout.equals(fromVideo.stdout)).toBe(true);
|
||||
}, 90000);
|
||||
|
||||
it.runIf(have)("refuses a video that genuinely has no audio track", async () => {
|
||||
const p = new LocalMusicProvider(dir);
|
||||
const silent = render("silent.mp4", [
|
||||
|
||||
+42
-9
@@ -56,11 +56,30 @@ const VIDEO_EXTENSIONS = new Set([
|
||||
".ogv",
|
||||
]);
|
||||
|
||||
/** Container the extracted audio track is remuxed into. Matroska takes
|
||||
/** Fallback container for an extracted audio track. Matroska takes
|
||||
* essentially any audio codec, so `-c:a copy` works without knowing what the
|
||||
* source used — no re-encode, no quality loss, no codec/extension table. */
|
||||
* source used — no re-encode, no codec/extension table. */
|
||||
const EXTRACTED_AUDIO_EXT = ".mka";
|
||||
|
||||
/**
|
||||
* Container to remux an extracted track into, chosen by its codec.
|
||||
*
|
||||
* AAC gets .m4a rather than the Matroska fallback. MP4 stores the AAC encoder
|
||||
* priming (the ~1000 warm-up samples every AAC encoder emits) in an edit list,
|
||||
* and that edit list does NOT survive into Matroska — so an aac→.mka remux
|
||||
* decodes ~23 ms longer than the source, with the priming samples audible at
|
||||
* the head instead of discarded. Measured: −66 dBFS, i.e. inaudible, but the
|
||||
* track is then fractionally out of step with its own reported duration for
|
||||
* no reason. Copying aac into .m4a keeps the edit list and decodes
|
||||
* byte-for-byte identical to the audio inside the original video.
|
||||
*
|
||||
* AAC is worth special-casing because it is what mp4 / mov / m4v — the
|
||||
* formats people actually upload — almost always carry.
|
||||
*/
|
||||
function extractedAudioExt(codec: string | null): string {
|
||||
return codec === "aac" ? ".m4a" : EXTRACTED_AUDIO_EXT;
|
||||
}
|
||||
|
||||
function isSupportedUploadExt(ext: string): boolean {
|
||||
return AUDIO_EXTENSIONS.has(ext) || VIDEO_EXTENSIONS.has(ext);
|
||||
}
|
||||
@@ -108,6 +127,9 @@ export interface MediaProbe {
|
||||
/** True when ffmpeg reported at least one audio stream. Only meaningful
|
||||
* together with `recognized` — see the comment there. */
|
||||
hasAudio: boolean;
|
||||
/** Lowercased codec name of the first audio stream ("aac", "mp3", "opus",
|
||||
* …), or null when there is none. Picks the remux container. */
|
||||
audioCodec: string | null;
|
||||
/**
|
||||
* True when ffmpeg actually opened the container and printed its
|
||||
* `Input #0, <format>, from '...'` header.
|
||||
@@ -138,11 +160,13 @@ export function parseMediaProbe(stderr: string): Omit<MediaProbe, "probed"> {
|
||||
// the bracketed id/language vary, so match on the "Audio:" tag itself. An
|
||||
// embedded cover image is a separate "Video: mjpeg ... [attached pic]" line
|
||||
// and never matches this.
|
||||
const hasAudio = /Stream #\d+:\d+[^\n]*:\s*Audio:/.test(stderr);
|
||||
const audioMatch = stderr.match(/Stream #\d+:\d+[^\n]*:\s*Audio:\s*([A-Za-z0-9_]+)/);
|
||||
const hasAudio = audioMatch !== null;
|
||||
const audioCodec = audioMatch ? audioMatch[1].toLowerCase() : null;
|
||||
// "Input #0, mov,mp4,m4a,3gp,3g2,mj2, from 'clip.mp4':" — absent entirely
|
||||
// when ffmpeg bails with "Error opening input: Invalid data found ...".
|
||||
const recognized = /^Input #\d+,/m.test(stderr);
|
||||
return { durationSeconds, hasAudio, recognized };
|
||||
return { durationSeconds, hasAudio, audioCodec, recognized };
|
||||
}
|
||||
|
||||
async function probeMedia(filePath: string): Promise<MediaProbe> {
|
||||
@@ -162,14 +186,14 @@ async function probeMedia(filePath: string): Promise<MediaProbe> {
|
||||
// slow, so allow more than the old 5s before giving up.
|
||||
const timeout = setTimeout(() => {
|
||||
ffmpeg.kill("SIGKILL");
|
||||
done({ durationSeconds: 0, hasAudio: false, recognized: false, probed: false });
|
||||
done({ durationSeconds: 0, hasAudio: false, audioCodec: null, recognized: false, probed: false });
|
||||
}, 20000);
|
||||
ffmpeg.stderr.on("data", (chunk) => {
|
||||
stderr += chunk.toString("utf8");
|
||||
});
|
||||
ffmpeg.on("error", () => {
|
||||
clearTimeout(timeout);
|
||||
done({ durationSeconds: 0, hasAudio: false, recognized: false, probed: false });
|
||||
done({ durationSeconds: 0, hasAudio: false, audioCodec: null, recognized: false, probed: false });
|
||||
});
|
||||
ffmpeg.on("close", () => {
|
||||
clearTimeout(timeout);
|
||||
@@ -309,7 +333,7 @@ export class LocalMusicProvider implements MusicProvider {
|
||||
try {
|
||||
probe = await probeMedia(filePath);
|
||||
} catch {
|
||||
probe = { durationSeconds: 0, hasAudio: false, recognized: false, probed: false };
|
||||
probe = { durationSeconds: 0, hasAudio: false, audioCodec: null, recognized: false, probed: false };
|
||||
}
|
||||
|
||||
// Reject a video with no audio track up front (#149). Left to playback it
|
||||
@@ -327,8 +351,17 @@ export class LocalMusicProvider implements MusicProvider {
|
||||
if (isVideo) {
|
||||
// Keep only the audio. The video bytes are dead weight against the
|
||||
// upload-directory quota and would never be used.
|
||||
const extracted = path.join(this.uploadDir, `${id}${EXTRACTED_AUDIO_EXT}`);
|
||||
if (await extractAudioTrack(filePath, extracted)) {
|
||||
// Preferred container first; if that remux fails (a codec the container
|
||||
// will not take), retry into Matroska, which takes almost anything.
|
||||
const preferredExt = extractedAudioExt(probe.audioCodec);
|
||||
let extracted = path.join(this.uploadDir, `${id}${preferredExt}`);
|
||||
let ok = await extractAudioTrack(filePath, extracted);
|
||||
if (!ok && preferredExt !== EXTRACTED_AUDIO_EXT) {
|
||||
rmSync(extracted, { force: true });
|
||||
extracted = path.join(this.uploadDir, `${id}${EXTRACTED_AUDIO_EXT}`);
|
||||
ok = await extractAudioTrack(filePath, extracted);
|
||||
}
|
||||
if (ok) {
|
||||
try {
|
||||
// Commit filePath and size TOGETHER, and only after the source is
|
||||
// actually gone. rmSync(force) still throws EBUSY/EPERM on Windows,
|
||||
|
||||
Reference in new issue
Block a user