mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-04 07:19:52 +00:00
feat(cli): export .srt/.vtt caption sidecars from a transcript (#1704)
Add formatSrt/formatVtt/wordsToCues to normalize.ts (the inverse of the existing parseSrt/parseVtt) and a 'hyperframes transcribe <transcript> --to srt|vtt' export mode. Word-level whisper transcripts group into cues on sentence boundaries with maxChars/maxGap guards; imported phrase-level cues pass through unchanged. Default transcribe behavior is unchanged and no new dependencies are added. Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>
This commit is contained in:
co-authored by
Matt Van Horn
parent
97db811a2f
commit
821bf2921a
@@ -1,5 +1,5 @@
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
|
||||
import { writeFileSync, mkdtempSync, rmSync } from "node:fs";
|
||||
import { writeFileSync, readFileSync, mkdtempSync, rmSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { tmpdir } from "node:os";
|
||||
import { WhisperUnavailableError } from "../whisper/manager.js";
|
||||
@@ -24,7 +24,7 @@ function dummyAudio(): { dir: string; input: string } {
|
||||
return { dir, input };
|
||||
}
|
||||
|
||||
describe("transcribe — whisper unavailable", () => {
|
||||
describe("transcribe command", () => {
|
||||
let dirs: string[] = [];
|
||||
let priorExitCode: typeof process.exitCode;
|
||||
|
||||
@@ -66,4 +66,60 @@ describe("transcribe — whisper unavailable", () => {
|
||||
expect(trackTranscribeUnavailable).toHaveBeenCalledWith({ optional: true });
|
||||
expect(trackCommandFailure).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("imports an SRT and exports an SRT sidecar from transcript.json", async () => {
|
||||
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
|
||||
dirs.push(dir);
|
||||
const input = join(dir, "sample.srt");
|
||||
const sample = `1
|
||||
00:00:01,000 --> 00:00:03,500
|
||||
Write HTML.
|
||||
|
||||
2
|
||||
00:00:03,500 --> 00:00:06,000
|
||||
Render video. Built for agents.
|
||||
`;
|
||||
writeFileSync(input, sample);
|
||||
|
||||
await transcribeCmd.run!({ args: { input, dir, json: true } } as never);
|
||||
const transcriptPath = join(dir, "transcript.json");
|
||||
|
||||
await transcribeCmd.run!({ args: { input: transcriptPath, to: "srt", json: true } } as never);
|
||||
const outputPath = join(dir, "transcript.srt");
|
||||
|
||||
expect(readFileSync(outputPath, "utf-8")).toBe(sample);
|
||||
const log = vi.mocked(console.log).mock.calls.at(-1)?.[0];
|
||||
expect(typeof log).toBe("string");
|
||||
if (typeof log !== "string") throw new Error("Expected JSON log output");
|
||||
expect(JSON.parse(log)).toEqual({
|
||||
ok: true,
|
||||
format: "srt",
|
||||
wordCount: 2,
|
||||
outputPath,
|
||||
});
|
||||
});
|
||||
|
||||
it("--preserve-cues keeps single-word cues separate when exporting from JSON", async () => {
|
||||
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
|
||||
dirs.push(dir);
|
||||
// Single-word cues have no internal whitespace, so the whitespace heuristic
|
||||
// can't tell them from word-level whisper output. --preserve-cues forces 1:1.
|
||||
const transcriptPath = join(dir, "transcript.json");
|
||||
writeFileSync(
|
||||
transcriptPath,
|
||||
JSON.stringify([
|
||||
{ text: "Yes", start: 0, end: 1 },
|
||||
{ text: "No", start: 1, end: 2 },
|
||||
]),
|
||||
);
|
||||
|
||||
await transcribeCmd.run!({
|
||||
args: { input: transcriptPath, to: "srt", "preserve-cues": true, json: true },
|
||||
} as never);
|
||||
|
||||
const output = readFileSync(join(dir, "transcript.srt"), "utf-8");
|
||||
expect(output).toBe(
|
||||
"1\n00:00:00,000 --> 00:00:01,000\nYes\n\n2\n00:00:01,000 --> 00:00:02,000\nNo\n",
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -2,6 +2,8 @@ import { defineCommand } from "citty";
|
||||
import type { Example } from "./_examples.js";
|
||||
import { existsSync, writeFileSync } from "node:fs";
|
||||
|
||||
type CaptionExportFormat = "srt" | "vtt";
|
||||
|
||||
export const examples: Example[] = [
|
||||
["Transcribe an audio file", "hyperframes transcribe audio.mp3"],
|
||||
["Transcribe a video file", "hyperframes transcribe video.mp4"],
|
||||
@@ -9,6 +11,11 @@ export const examples: Example[] = [
|
||||
["Set language to filter non-target speech", "hyperframes transcribe audio.mp3 --language en"],
|
||||
["Import an existing SRT file", "hyperframes transcribe subtitles.srt"],
|
||||
["Import an OpenAI Whisper JSON response", "hyperframes transcribe response.json"],
|
||||
["Export captions to SRT", "hyperframes transcribe transcript.json --to srt"],
|
||||
[
|
||||
"Export single-word/CJK captions without re-grouping",
|
||||
"hyperframes transcribe transcript.json --to vtt --preserve-cues",
|
||||
],
|
||||
];
|
||||
import { resolve, join, extname, dirname } from "node:path";
|
||||
import * as clack from "@clack/prompts";
|
||||
@@ -49,6 +56,21 @@ export default defineCommand({
|
||||
description: "Output result as JSON",
|
||||
default: false,
|
||||
},
|
||||
to: {
|
||||
type: "string",
|
||||
description: "Export transcript sidecar format: srt or vtt",
|
||||
},
|
||||
output: {
|
||||
type: "string",
|
||||
alias: "o",
|
||||
description: "Output path for exported SRT/VTT sidecar",
|
||||
},
|
||||
"preserve-cues": {
|
||||
type: "boolean",
|
||||
description:
|
||||
"Keep each transcript entry as its own caption cue (skip word-level grouping). Use when exporting an already-cued transcript whose entries have no internal spaces, e.g. single-word or CJK captions.",
|
||||
default: false,
|
||||
},
|
||||
optional: {
|
||||
type: "boolean",
|
||||
description:
|
||||
@@ -73,6 +95,17 @@ export default defineCommand({
|
||||
|
||||
// ── Import mode: convert existing transcript ──────────────────────────
|
||||
const isImport = ext === ".json" || ext === ".srt" || ext === ".vtt";
|
||||
const to = parseExportFormat(args.to, args.json);
|
||||
|
||||
if (to) {
|
||||
if (!isImport) {
|
||||
failWith(
|
||||
"--to can only export from transcript files (.json, .srt, .vtt). Run transcribe first.",
|
||||
args.json,
|
||||
);
|
||||
}
|
||||
return exportTranscript(inputPath, dir, to, args.output, args.json, args["preserve-cues"]);
|
||||
}
|
||||
|
||||
if (isImport) {
|
||||
return importTranscript(inputPath, dir, args.json);
|
||||
@@ -88,20 +121,40 @@ export default defineCommand({
|
||||
},
|
||||
});
|
||||
|
||||
function failWith(message: string, json: boolean): never {
|
||||
trackCommandFailure("transcribe", message);
|
||||
if (json) {
|
||||
console.log(JSON.stringify({ ok: false, error: message }));
|
||||
} else {
|
||||
console.error(c.error(message));
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
function parseExportFormat(
|
||||
value: string | undefined,
|
||||
json: boolean,
|
||||
): CaptionExportFormat | undefined {
|
||||
if (!value) return undefined;
|
||||
const normalized = value.toLowerCase();
|
||||
if (normalized === "srt" || normalized === "vtt") return normalized;
|
||||
|
||||
failWith(`Unsupported caption export format: ${value}. Use srt or vtt.`, json);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Import existing transcript
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
function exitNoWords(json: boolean): never {
|
||||
failWith("No words found in transcript.", json);
|
||||
}
|
||||
|
||||
async function importTranscript(inputPath: string, dir: string, json: boolean): Promise<void> {
|
||||
const { loadTranscript, patchCaptionHtml } = await import("../whisper/normalize.js");
|
||||
const { words, format } = loadTranscript(inputPath);
|
||||
|
||||
if (words.length === 0) {
|
||||
const message = "No words found in transcript.";
|
||||
trackCommandFailure("transcribe", message);
|
||||
console.error(c.error(message));
|
||||
process.exit(1);
|
||||
}
|
||||
if (words.length === 0) exitNoWords(json);
|
||||
|
||||
const outPath = join(dir, "transcript.json");
|
||||
writeFileSync(outPath, JSON.stringify(words, null, 2));
|
||||
@@ -118,6 +171,44 @@ async function importTranscript(inputPath: string, dir: string, json: boolean):
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Export transcript sidecars
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
async function exportTranscript(
|
||||
inputPath: string,
|
||||
dir: string,
|
||||
to: CaptionExportFormat,
|
||||
output: string | undefined,
|
||||
json: boolean,
|
||||
preserveCues: boolean,
|
||||
): Promise<void> {
|
||||
const { loadTranscript, formatSrt, formatVtt } = await import("../whisper/normalize.js");
|
||||
const { words, format } = loadTranscript(inputPath);
|
||||
|
||||
if (words.length === 0) exitNoWords(json);
|
||||
|
||||
// A .srt/.vtt source is already phrase-level; keep its cue boundaries 1:1.
|
||||
// --preserve-cues forces the same for an already-cued transcript.json whose
|
||||
// entries have no internal whitespace (single-word or CJK captions), which
|
||||
// the automatic whitespace heuristic in wordsToCues can't detect.
|
||||
const preGrouped = preserveCues || format === "srt" || format === "vtt" || undefined;
|
||||
const outPath = resolve(output ?? join(dir, `transcript.${to}`));
|
||||
const content =
|
||||
to === "srt" ? formatSrt(words, { preGrouped }) : formatVtt(words, { preGrouped });
|
||||
writeFileSync(outPath, content);
|
||||
|
||||
if (json) {
|
||||
console.log(
|
||||
JSON.stringify({ ok: true, format: to, wordCount: words.length, outputPath: outPath }),
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`${c.success("◇")} Exported ${c.accent(String(words.length))} words to ${c.accent(to.toUpperCase())} → ${c.accent(outPath)}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Transcribe audio/video with whisper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user