mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-03 04:38:33 +00:00
perf(engine): dedupe identical extractions within one render (#1900)
* perf(engine): write PNG frames at compression_level 1 Extracted video frames are render-scoped temp files read once during capture, so zlib effort above level 1 buys nothing. Measured 3.3x faster on 60s of 1080p H.264 to PNG (11.4s to 3.5s) and 5.4x on a 20s vp9-alpha webm (4.3s to 0.79s), for ~14% larger temp files. * perf(engine): one-pass VFR extraction with -fps_mode cfr VFR sources (screen recordings, phone videos) were re-encoded to CFR with libx264 and then extracted in a second ffmpeg pass. Extraction now runs a single pass with -fps_mode cfr -r <fps>. Same frame counts on the VFR regression fixtures (120/120 mid-seek, 297-303 full file), one less x264 generation of quality loss, ~3.4x faster on VFR inputs. convertVfrToCfr and the _vfr_normalized intermediate are deleted. The full-VFR test's byte-identical duplicate-frame cap is retired with cause: the fixture has no source frames for 40% of its timeline, so held frames are correct; the two-pass path only scored under it because x264 encoder noise made frozen frames hash differently. The freeze regression (missing frames) stays pinned by the frame-count windows. * docs(engine): pin vfrPreflightMs definition change after one-pass VFR vfrPreflightMs used to time a per-source VFR-to-CFR re-encode; it now times only the cached classification probe and collapses to ~0. Call that out on ExtractionPhaseBreakdown so dashboards keyed on the old threshold semantics migrate to vfrPreflightCount / extractMs. * fix(engine): bump extraction cache schema to v3 for one-pass VFR frames One-pass VFR extraction changes frame CONTENTS for VFR sources while the cache key tuple (path, mtime, size, trim, fps, format) is unchanged, so warm v2 entries holding two-pass frames would keep being served across the deploy boundary. Bumping the schema prefix makes v2 entries inert; affected sources re-extract once. * perf(engine): dedupe identical extractions within one render N <video> elements sharing (resolved path, mediaStart, duration, fps, format) extracted N times; they now share one extraction via an in-flight promise map keyed on that tuple. Duplicate elements receive the shared frame set under their own videoId. This also removes a race where two identical clips on a cache miss wrote the same extraction-cache entry dir concurrently. 3x duplicated 60s 1080p video: 4426ms to 1521ms in the A/B benchmark, one frame set on disk. * fix(engine): attribute shared-extraction failures to the dedupe leader When a deduped extraction fails, every follower reported the leader's error verbatim under its own videoId, reading as N independent failures in traces. Follower errors now carry a '[shared extraction, leader <id>]' prefix so the fan-out is traceable to one root failure.
This commit is contained in:
@@ -682,6 +682,36 @@ describe.skipIf(!HAS_FFMPEG)("video frame extraction format", () => {
|
||||
rmSync(cacheDir, { recursive: true, force: true });
|
||||
}
|
||||
}, 60_000);
|
||||
|
||||
it("dedupes identical extractions within one render", async () => {
|
||||
const outputDir = join(FIXTURE_DIR, "out-dedupe");
|
||||
mkdirSync(outputDir, { recursive: true });
|
||||
|
||||
const videoA: VideoElement = { ...fixtureVideo(), id: "dupe-a" };
|
||||
const videoB: VideoElement = { ...fixtureVideo(), id: "dupe-b" };
|
||||
|
||||
const result = await extractAllVideoFrames([videoA, videoB], FIXTURE_DIR, {
|
||||
fps: 1,
|
||||
outputDir,
|
||||
});
|
||||
|
||||
expect(result.errors).toEqual([]);
|
||||
expect(result.extracted).toHaveLength(2);
|
||||
const first = result.extracted[0]!;
|
||||
const second = result.extracted[1]!;
|
||||
expect(first.videoId).toBe("dupe-a");
|
||||
expect(second.videoId).toBe("dupe-b");
|
||||
expect(second.outputDir).toBe(first.outputDir);
|
||||
expect(Array.from(second.framePaths.entries())).toEqual(Array.from(first.framePaths.entries()));
|
||||
|
||||
const frameDirs = readdirSync(outputDir, { withFileTypes: true })
|
||||
.filter((entry) => entry.isDirectory())
|
||||
.map((entry) => entry.name);
|
||||
expect(frameDirs).toEqual(["dupe-a"]);
|
||||
expect(readdirSync(first.outputDir).filter((f) => f.endsWith(".jpg"))).toHaveLength(
|
||||
first.totalFrames,
|
||||
);
|
||||
}, 60_000);
|
||||
});
|
||||
|
||||
// Regression test for the VFR (variable frame rate) freeze bug.
|
||||
|
||||
@@ -746,17 +746,17 @@ export async function extractAllVideoFrames(
|
||||
videoPath: string,
|
||||
videoDuration: number,
|
||||
i: number,
|
||||
metadata: VideoMetadata,
|
||||
cacheFormat: CacheFrameFormat,
|
||||
): Promise<ExtractedFrames | null> {
|
||||
if (!cacheRootDir) return null;
|
||||
const keyInput = cacheKeyInputs[i];
|
||||
const probedMeta = videoMetadata[i];
|
||||
if (!keyInput || !probedMeta) return null;
|
||||
const cacheFormat = resolveFrameFormat(probedMeta, options.format);
|
||||
if (!keyInput) return null;
|
||||
|
||||
const keyDuration = resolveSegmentDuration(
|
||||
keyInput.end - keyInput.start,
|
||||
keyInput.mediaStart,
|
||||
probedMeta,
|
||||
metadata,
|
||||
);
|
||||
const lookup = lookupCacheEntry(cacheRootDir, {
|
||||
videoPath: keyInput.videoPath,
|
||||
@@ -775,7 +775,7 @@ export async function extractAllVideoFrames(
|
||||
srcPath: keyInput.videoPath,
|
||||
fps: options.fps,
|
||||
format: cacheFormat,
|
||||
metadata: probedMeta,
|
||||
metadata,
|
||||
});
|
||||
return { ...rehydrated, ownedByLookup: true };
|
||||
}
|
||||
@@ -798,43 +798,118 @@ export async function extractAllVideoFrames(
|
||||
return { ...result, ownedByLookup: true };
|
||||
}
|
||||
|
||||
const results = await Promise.all(
|
||||
resolvedVideos.map(async ({ video, videoPath }, i) => {
|
||||
function extractionError(videoId: string, err: unknown): { videoId: string; error: string } {
|
||||
return { videoId, error: err instanceof Error ? err.message : String(err) };
|
||||
}
|
||||
|
||||
type PreparedExtraction = {
|
||||
video: VideoElement;
|
||||
videoPath: string;
|
||||
index: number;
|
||||
metadata: VideoMetadata;
|
||||
videoDuration: number;
|
||||
format: CacheFrameFormat;
|
||||
dedupeKey: string;
|
||||
};
|
||||
|
||||
type PreparedExtractionResult =
|
||||
| { work: PreparedExtraction }
|
||||
| { error: { videoId: string; error: string } };
|
||||
|
||||
const preparedExtractions: PreparedExtractionResult[] = await Promise.all(
|
||||
resolvedVideos.map(async ({ video, videoPath }, index) => {
|
||||
if (signal?.aborted) {
|
||||
throw new Error("Video frame extraction cancelled");
|
||||
}
|
||||
try {
|
||||
const probedMeta = videoMetadata[i] ?? (await extractMediaMetadata(videoPath));
|
||||
const metadata = videoMetadata[index] ?? (await extractMediaMetadata(videoPath));
|
||||
const videoDuration = resolveSegmentDuration(
|
||||
video.end - video.start,
|
||||
video.mediaStart,
|
||||
probedMeta,
|
||||
metadata,
|
||||
);
|
||||
if (video.end - video.start !== videoDuration) {
|
||||
video.end = video.start + videoDuration;
|
||||
}
|
||||
|
||||
const cached = await tryCachedExtract(video, videoPath, videoDuration, i);
|
||||
if (cached) return { result: cached };
|
||||
const format = resolveFrameFormat(metadata, options.format);
|
||||
const dedupeKey = `${videoPath}\0${video.mediaStart}\0${videoDuration}\0${options.fps}\0${format}`;
|
||||
|
||||
const result = await extractVideoFramesRange(
|
||||
return {
|
||||
work: {
|
||||
video,
|
||||
videoPath,
|
||||
video.id,
|
||||
video.mediaStart,
|
||||
index,
|
||||
metadata,
|
||||
videoDuration,
|
||||
{ ...options, format: resolveFrameFormat(probedMeta, options.format) },
|
||||
format,
|
||||
dedupeKey,
|
||||
},
|
||||
};
|
||||
} catch (err) {
|
||||
return { error: extractionError(video.id, err) };
|
||||
}
|
||||
}),
|
||||
);
|
||||
|
||||
// Value carries the leader's videoId so a shared-extraction failure can be
|
||||
// attributed: N deduped elements otherwise report the same root error under
|
||||
// N different videoIds, which reads as N independent failures in traces.
|
||||
const inFlightExtractions = new Map<
|
||||
string,
|
||||
{ leaderVideoId: string; promise: Promise<ExtractedFrames> }
|
||||
>();
|
||||
const results = await Promise.all(
|
||||
preparedExtractions.map(async (prepared) => {
|
||||
if ("error" in prepared) return prepared;
|
||||
const { work } = prepared;
|
||||
|
||||
try {
|
||||
const existing = inFlightExtractions.get(work.dedupeKey);
|
||||
if (existing) {
|
||||
try {
|
||||
const shared = await existing.promise;
|
||||
return { result: { ...shared, videoId: work.video.id } };
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
return {
|
||||
error: {
|
||||
videoId: work.video.id,
|
||||
error: `[shared extraction, leader ${existing.leaderVideoId}] ${message}`,
|
||||
},
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
const extraction = (async () => {
|
||||
const cached = await tryCachedExtract(
|
||||
work.video,
|
||||
work.videoPath,
|
||||
work.videoDuration,
|
||||
work.index,
|
||||
work.metadata,
|
||||
work.format,
|
||||
);
|
||||
if (cached) return cached;
|
||||
|
||||
return extractVideoFramesRange(
|
||||
work.videoPath,
|
||||
work.video.id,
|
||||
work.video.mediaStart,
|
||||
work.videoDuration,
|
||||
{ ...options, format: work.format },
|
||||
signal,
|
||||
config,
|
||||
);
|
||||
})();
|
||||
|
||||
return { result };
|
||||
inFlightExtractions.set(work.dedupeKey, {
|
||||
leaderVideoId: work.video.id,
|
||||
promise: extraction,
|
||||
});
|
||||
return { result: await extraction };
|
||||
} catch (err) {
|
||||
return {
|
||||
error: {
|
||||
videoId: video.id,
|
||||
error: err instanceof Error ? err.message : String(err),
|
||||
},
|
||||
};
|
||||
return { error: extractionError(work.video.id, err) };
|
||||
}
|
||||
}),
|
||||
);
|
||||
|
||||
Reference in New Issue
Block a user