mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-03 04:38:33 +00:00
Merge pull request #939 from heygen-com/feat/auto-size-chunk-size-when-undefined
feat(producer): auto-size chunkSize from maxParallelChunks when undefined
This commit is contained in:
@@ -44,6 +44,7 @@ export {
|
|||||||
// Constants
|
// Constants
|
||||||
DEFAULT_CHUNK_SIZE,
|
DEFAULT_CHUNK_SIZE,
|
||||||
DEFAULT_MAX_PARALLEL_CHUNKS,
|
DEFAULT_MAX_PARALLEL_CHUNKS,
|
||||||
|
MIN_CHUNK_SIZE,
|
||||||
PLAN_DIR_SIZE_LIMIT_BYTES,
|
PLAN_DIR_SIZE_LIMIT_BYTES,
|
||||||
PLAN_PROJECT_DIR_SKIP_SEGMENTS,
|
PLAN_PROJECT_DIR_SKIP_SEGMENTS,
|
||||||
// Error codes + classes
|
// Error codes + classes
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ import {
|
|||||||
buildChunkSlices,
|
buildChunkSlices,
|
||||||
DEFAULT_CHUNK_SIZE,
|
DEFAULT_CHUNK_SIZE,
|
||||||
DEFAULT_MAX_PARALLEL_CHUNKS,
|
DEFAULT_MAX_PARALLEL_CHUNKS,
|
||||||
|
MIN_CHUNK_SIZE,
|
||||||
plan,
|
plan,
|
||||||
resolveChunkPlan,
|
resolveChunkPlan,
|
||||||
} from "./plan.js";
|
} from "./plan.js";
|
||||||
@@ -93,6 +94,37 @@ describe("resolveChunkPlan", () => {
|
|||||||
expect(() => resolveChunkPlan(60, 240, 16.5)).toThrow(/positive integer/);
|
expect(() => resolveChunkPlan(60, 240, 16.5)).toThrow(/positive integer/);
|
||||||
expect(() => resolveChunkPlan(60, 240, Number.POSITIVE_INFINITY)).toThrow(/positive integer/);
|
expect(() => resolveChunkPlan(60, 240, Number.POSITIVE_INFINITY)).toThrow(/positive integer/);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// ── Auto-size when configChunkSize is undefined ───────────────────────
|
||||||
|
// The auto-sizer picks `max(MIN_CHUNK_SIZE, ceil(totalFrames /
|
||||||
|
// maxParallelChunks))` whenever the caller leaves `chunkSize` undefined,
|
||||||
|
// honoring `maxParallelChunks` instead of clamping at a 240-frame default.
|
||||||
|
|
||||||
|
it("explicit chunkSize wins: 660 frames + chunkSize=240 + maxParallelChunks=16 → 3 chunks", () => {
|
||||||
|
// Regression guard for the "explicit number still works" half of the
|
||||||
|
// contract — passing 240 explicitly must not get auto-sized.
|
||||||
|
const result = resolveChunkPlan(660, 240, 16);
|
||||||
|
expect(result.chunkCount).toBe(3);
|
||||||
|
expect(result.effectiveChunkSize).toBe(240);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("auto-sizes when chunkSize=undefined: 660 frames + maxParallelChunks=16 → 16 chunks", () => {
|
||||||
|
// ceil(660 / 16) = 42; max(MIN_CHUNK_SIZE=10, 42) = 42. naiveCount =
|
||||||
|
// ceil(660 / 42) = 16, which lands exactly at the cap.
|
||||||
|
const result = resolveChunkPlan(660, undefined, 16);
|
||||||
|
expect(result.chunkCount).toBe(16);
|
||||||
|
expect(result.effectiveChunkSize).toBe(42);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("auto-size floor: tiny renders cap at MIN_CHUNK_SIZE rather than fragmenting infinitely", () => {
|
||||||
|
// 50 frames / 16 workers naively gives a 4-frame chunk size, which
|
||||||
|
// would produce 13 chunks of 4 frames each — per-chunk fixed overhead
|
||||||
|
// dwarfs the parallelism gain. The MIN_CHUNK_SIZE=10 floor pins
|
||||||
|
// chunkSize at 10, producing ceil(50/10) = 5 chunks instead.
|
||||||
|
const result = resolveChunkPlan(50, undefined, 16);
|
||||||
|
expect(result.chunkCount).toBe(5);
|
||||||
|
expect(result.effectiveChunkSize).toBe(MIN_CHUNK_SIZE);
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("buildChunkSlices", () => {
|
describe("buildChunkSlices", () => {
|
||||||
@@ -116,6 +148,7 @@ describe("plan() defaults", () => {
|
|||||||
it("exports the documented chunking defaults", () => {
|
it("exports the documented chunking defaults", () => {
|
||||||
expect(DEFAULT_CHUNK_SIZE).toBe(240);
|
expect(DEFAULT_CHUNK_SIZE).toBe(240);
|
||||||
expect(DEFAULT_MAX_PARALLEL_CHUNKS).toBe(16);
|
expect(DEFAULT_MAX_PARALLEL_CHUNKS).toBe(16);
|
||||||
|
expect(MIN_CHUNK_SIZE).toBe(10);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -131,9 +164,12 @@ describe("plan() — golden planDir + planHash determinism", () => {
|
|||||||
async () => {
|
async () => {
|
||||||
const planDir = join(runRoot, "plan-layout");
|
const planDir = join(runRoot, "plan-layout");
|
||||||
mkdirSync(planDir, { recursive: true });
|
mkdirSync(planDir, { recursive: true });
|
||||||
|
// Pin chunkSize=240 so this fixture exercises the single-chunk path
|
||||||
|
// (totalFrames=30 → ceil(30/240)=1 chunk). The auto-sized variant
|
||||||
|
// (chunkSize=undefined) is exercised by the dedicated test below.
|
||||||
const result = await plan(
|
const result = await plan(
|
||||||
projectDir,
|
projectDir,
|
||||||
{ fps: 30, width: 320, height: 240, format: "mp4" },
|
{ fps: 30, width: 320, height: 240, format: "mp4", chunkSize: 240 },
|
||||||
planDir,
|
planDir,
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -152,7 +188,7 @@ describe("plan() — golden planDir + planHash determinism", () => {
|
|||||||
// ── PlanResult contract ─────────────────────────────────────────────
|
// ── PlanResult contract ─────────────────────────────────────────────
|
||||||
expect(result.planDir).toBe(planDir);
|
expect(result.planDir).toBe(planDir);
|
||||||
expect(result.planHash).toMatch(/^[0-9a-f]{64}$/);
|
expect(result.planHash).toMatch(/^[0-9a-f]{64}$/);
|
||||||
expect(result.chunkCount).toBeGreaterThanOrEqual(1);
|
expect(result.chunkCount).toBe(1);
|
||||||
expect(result.totalFrames).toBe(30); // 1s @ 30fps
|
expect(result.totalFrames).toBe(30); // 1s @ 30fps
|
||||||
expect(result.width).toBe(320);
|
expect(result.width).toBe(320);
|
||||||
expect(result.height).toBe(240);
|
expect(result.height).toBe(240);
|
||||||
@@ -185,6 +221,37 @@ describe("plan() — golden planDir + planHash determinism", () => {
|
|||||||
TIMEOUT_MS,
|
TIMEOUT_MS,
|
||||||
);
|
);
|
||||||
|
|
||||||
|
it(
|
||||||
|
"auto-sizes chunkSize end-to-end when caller omits it",
|
||||||
|
async () => {
|
||||||
|
// Integration check that the auto-sizer wired through plan() actually
|
||||||
|
// produces multi-chunk output for the same fixture that single-chunks
|
||||||
|
// when chunkSize is pinned. With totalFrames=30 and the default
|
||||||
|
// maxParallelChunks=16, the auto-sizer picks
|
||||||
|
// max(MIN_CHUNK_SIZE=10, ceil(30/16)=2) = 10 → ceil(30/10) = 3 chunks.
|
||||||
|
const planDir = join(runRoot, "plan-autosized");
|
||||||
|
mkdirSync(planDir, { recursive: true });
|
||||||
|
const result = await plan(
|
||||||
|
projectDir,
|
||||||
|
{ fps: 30, width: 320, height: 240, format: "mp4" },
|
||||||
|
planDir,
|
||||||
|
);
|
||||||
|
expect(result.chunkCount).toBe(3);
|
||||||
|
const chunks = JSON.parse(
|
||||||
|
readFileSync(join(planDir, "meta", "chunks.json"), "utf-8"),
|
||||||
|
) as Array<{ index: number; startFrame: number; endFrame: number }>;
|
||||||
|
expect(chunks).toHaveLength(3);
|
||||||
|
// Encoder gopSize must follow the auto-sized chunk so chunk-boundary
|
||||||
|
// IDR keyframes still land at frame 0 of each chunk.
|
||||||
|
const encoder = JSON.parse(
|
||||||
|
readFileSync(join(planDir, "meta", "encoder.json"), "utf-8"),
|
||||||
|
) as Record<string, unknown>;
|
||||||
|
expect(encoder.gopSize).toBe(10);
|
||||||
|
expect(encoder.chunkSize).toBe(10);
|
||||||
|
},
|
||||||
|
TIMEOUT_MS,
|
||||||
|
);
|
||||||
|
|
||||||
it(
|
it(
|
||||||
"produces a byte-identical planHash on a second invocation",
|
"produces a byte-identical planHash on a second invocation",
|
||||||
async () => {
|
async () => {
|
||||||
|
|||||||
@@ -99,7 +99,24 @@ export interface DistributedRenderConfig {
|
|||||||
/** Output resolution preset; engages Chrome `deviceScaleFactor` supersampling. */
|
/** Output resolution preset; engages Chrome `deviceScaleFactor` supersampling. */
|
||||||
outputResolution?: CanvasResolution;
|
outputResolution?: CanvasResolution;
|
||||||
|
|
||||||
/** Default `240` frames (~8s @ 30fps; fits Lambda's 15-min cap). */
|
/**
|
||||||
|
* Frames per chunk. When explicitly set, that value is used and
|
||||||
|
* `chunkCount = min(maxParallelChunks, ceil(totalFrames / chunkSize))`
|
||||||
|
* — useful when the caller wants a specific per-chunk runtime
|
||||||
|
* regardless of fan-out. When `undefined` (the default), `plan()`
|
||||||
|
* auto-sizes from `maxParallelChunks` so the caller's fan-out
|
||||||
|
* intent is honored: `effectiveChunkSize = max(MIN_CHUNK_SIZE,
|
||||||
|
* ceil(totalFrames / maxParallelChunks))`. The auto-size floor
|
||||||
|
* (`MIN_CHUNK_SIZE = 10`) keeps per-chunk fixed overhead from
|
||||||
|
* swamping the parallelism gain on tiny renders.
|
||||||
|
*
|
||||||
|
* `effectiveChunkSize` also drives `LockedRenderConfig.gopSize` — every
|
||||||
|
* chunk's first frame is an IDR keyframe, so smaller chunks mean a
|
||||||
|
* tighter GOP and larger encoded files. Callers who optimize for
|
||||||
|
* output bytes (rather than wall-clock parallelism) should pass an
|
||||||
|
* explicit `chunkSize` matching their target GOP — e.g. `240` for the
|
||||||
|
* old 8-second-GOP behavior.
|
||||||
|
*/
|
||||||
chunkSize?: number;
|
chunkSize?: number;
|
||||||
/** Default `16`. Caps long renders to fewer-but-longer chunks for operational fairness. */
|
/** Default `16`. Caps long renders to fewer-but-longer chunks for operational fairness. */
|
||||||
maxParallelChunks?: number;
|
maxParallelChunks?: number;
|
||||||
@@ -177,10 +194,21 @@ export const PLAN_PROJECT_DIR_SKIP_SEGMENTS: ReadonlySet<string> = new Set([
|
|||||||
".turbo",
|
".turbo",
|
||||||
]);
|
]);
|
||||||
|
|
||||||
/** Default chunk size in frames (~8s @ 30fps; fits Lambda's 15-min cap). */
|
/**
|
||||||
|
* Default chunk size in frames (~8s @ 30fps; fits Lambda's 15-min cap).
|
||||||
|
* Used when the caller explicitly passes this value. When `chunkSize` is
|
||||||
|
* `undefined`, `plan()` auto-sizes from `maxParallelChunks` instead.
|
||||||
|
*/
|
||||||
export const DEFAULT_CHUNK_SIZE = 240;
|
export const DEFAULT_CHUNK_SIZE = 240;
|
||||||
/** Default cap on parallel chunks for operational fairness across renders. */
|
/** Default cap on parallel chunks for operational fairness across renders. */
|
||||||
export const DEFAULT_MAX_PARALLEL_CHUNKS = 16;
|
export const DEFAULT_MAX_PARALLEL_CHUNKS = 16;
|
||||||
|
/**
|
||||||
|
* Floor for the auto-sized `chunkSize` when the caller leaves it
|
||||||
|
* `undefined`. Anything smaller hits a per-chunk fixed-overhead wall
|
||||||
|
* (worker boot + plan download + planHash recompute + ffmpeg init) that
|
||||||
|
* outweighs the parallelism gain on tiny renders.
|
||||||
|
*/
|
||||||
|
export const MIN_CHUNK_SIZE = 10;
|
||||||
/**
|
/**
|
||||||
* Default hard ceiling on `<planDir>/` size in bytes. 2 GB fits inside
|
* Default hard ceiling on `<planDir>/` size in bytes. 2 GB fits inside
|
||||||
* AWS Lambda's 10 GB `/tmp` alongside the chunk worker's captured frames
|
* AWS Lambda's 10 GB `/tmp` alongside the chunk worker's captured frames
|
||||||
@@ -329,18 +357,26 @@ export function measurePlanDirBytes(planDir: string): number {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Compute `(chunkCount, effectiveChunkSize)` from total frames and the
|
* Compute `(chunkCount, effectiveChunkSize)` from total frames and the
|
||||||
* caller's chunking knobs:
|
* caller's chunking knobs. The operative chunk size is
|
||||||
|
* `resolvedChunkSize` — equal to `configChunkSize` when the caller
|
||||||
|
* passes one, otherwise auto-sized from `maxParallelChunks`:
|
||||||
*
|
*
|
||||||
* chunkCount = min(maxParallelChunks, ceil(totalFrames / chunkSize))
|
* resolvedChunkSize = configChunkSize ?? max(MIN_CHUNK_SIZE, ceil(totalFrames / maxParallelChunks))
|
||||||
* effectiveChunkSize = max(configChunkSize, ceil(totalFrames / maxParallelChunks))
|
* chunkCount = min(maxParallelChunks, ceil(totalFrames / resolvedChunkSize))
|
||||||
|
* effectiveChunkSize = max(resolvedChunkSize, ceil(totalFrames / chunkCount))
|
||||||
*
|
*
|
||||||
* Long renders auto-rescale to fewer-but-longer chunks rather than
|
* Long renders auto-rescale to fewer-but-longer chunks rather than
|
||||||
* fragmenting infinitely. Returned `chunkCount >= 1` (`totalFrames === 0`
|
* fragmenting infinitely. Returned `chunkCount >= 1` (`totalFrames === 0`
|
||||||
* is rejected upstream); `effectiveChunkSize >= configChunkSize`.
|
* is rejected upstream); `effectiveChunkSize >= resolvedChunkSize`.
|
||||||
|
*
|
||||||
|
* The auto-sizer (triggered when `configChunkSize` is `undefined`) honors
|
||||||
|
* the caller's fan-out intent: passing `maxParallelChunks=16` without
|
||||||
|
* `chunkSize` produces 16 chunks (subject to the `MIN_CHUNK_SIZE` floor
|
||||||
|
* on tiny renders). Explicit numbers, including `240`, take precedence.
|
||||||
*/
|
*/
|
||||||
export function resolveChunkPlan(
|
export function resolveChunkPlan(
|
||||||
totalFrames: number,
|
totalFrames: number,
|
||||||
configChunkSize: number,
|
configChunkSize: number | undefined,
|
||||||
maxParallelChunks: number,
|
maxParallelChunks: number,
|
||||||
): { chunkCount: number; effectiveChunkSize: number } {
|
): { chunkCount: number; effectiveChunkSize: number } {
|
||||||
// Integer-only inputs: a fractional `totalFrames` (e.g. 10.5) would
|
// Integer-only inputs: a fractional `totalFrames` (e.g. 10.5) would
|
||||||
@@ -348,11 +384,20 @@ export function resolveChunkPlan(
|
|||||||
// chunk worker's `for (i = startFrame; i < endFrame; i++)` loop would
|
// chunk worker's `for (i = startFrame; i < endFrame; i++)` loop would
|
||||||
// silently truncate.
|
// silently truncate.
|
||||||
assertPositiveInteger("totalFrames", totalFrames);
|
assertPositiveInteger("totalFrames", totalFrames);
|
||||||
assertPositiveInteger("configChunkSize", configChunkSize);
|
|
||||||
assertPositiveInteger("maxParallelChunks", maxParallelChunks);
|
assertPositiveInteger("maxParallelChunks", maxParallelChunks);
|
||||||
const naiveCount = Math.ceil(totalFrames / configChunkSize);
|
// Validate the caller-supplied value with its real name so the error
|
||||||
|
// message points at the actual bad input. The auto-sized branch is
|
||||||
|
// provably a positive integer (totalFrames and maxParallelChunks are
|
||||||
|
// already validated above, MIN_CHUNK_SIZE is a positive integer
|
||||||
|
// constant), so it doesn't need re-checking.
|
||||||
|
if (configChunkSize !== undefined) {
|
||||||
|
assertPositiveInteger("configChunkSize", configChunkSize);
|
||||||
|
}
|
||||||
|
const resolvedChunkSize =
|
||||||
|
configChunkSize ?? Math.max(MIN_CHUNK_SIZE, Math.ceil(totalFrames / maxParallelChunks));
|
||||||
|
const naiveCount = Math.ceil(totalFrames / resolvedChunkSize);
|
||||||
const chunkCount = Math.min(maxParallelChunks, Math.max(1, naiveCount));
|
const chunkCount = Math.min(maxParallelChunks, Math.max(1, naiveCount));
|
||||||
const effectiveChunkSize = Math.max(configChunkSize, Math.ceil(totalFrames / chunkCount));
|
const effectiveChunkSize = Math.max(resolvedChunkSize, Math.ceil(totalFrames / chunkCount));
|
||||||
return { chunkCount, effectiveChunkSize };
|
return { chunkCount, effectiveChunkSize };
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -745,11 +790,10 @@ export async function plan(
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ── Chunking decisions + locked config ──
|
// ── Chunking decisions + locked config ──
|
||||||
const configChunkSize = config.chunkSize ?? DEFAULT_CHUNK_SIZE;
|
|
||||||
const maxParallel = config.maxParallelChunks ?? DEFAULT_MAX_PARALLEL_CHUNKS;
|
const maxParallel = config.maxParallelChunks ?? DEFAULT_MAX_PARALLEL_CHUNKS;
|
||||||
const { chunkCount, effectiveChunkSize } = resolveChunkPlan(
|
const { chunkCount, effectiveChunkSize } = resolveChunkPlan(
|
||||||
totalFrames,
|
totalFrames,
|
||||||
configChunkSize,
|
config.chunkSize,
|
||||||
maxParallel,
|
maxParallel,
|
||||||
);
|
);
|
||||||
const chunks = buildChunkSlices(totalFrames, chunkCount, effectiveChunkSize);
|
const chunks = buildChunkSlices(totalFrames, chunkCount, effectiveChunkSize);
|
||||||
|
|||||||
Reference in New Issue
Block a user