feat(cli): export .srt/.vtt caption sidecars from a transcript (#1704)

Add formatSrt/formatVtt/wordsToCues to normalize.ts (the inverse of the
existing parseSrt/parseVtt) and a 'hyperframes transcribe <transcript> --to
srt|vtt' export mode. Word-level whisper transcripts group into cues on
sentence boundaries with maxChars/maxGap guards; imported phrase-level cues
pass through unchanged. Default transcribe behavior is unchanged and no new
dependencies are added.

Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>
This commit is contained in:
Matt Van Horn
2026-06-24 19:03:16 -04:00
committed by GitHub
co-authored by Matt Van Horn
parent 97db811a2f
commit 821bf2921a
5 changed files with 428 additions and 11 deletions
+58 -2
View File
@@ -1,5 +1,5 @@
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
import { writeFileSync, mkdtempSync, rmSync } from "node:fs";
import { writeFileSync, readFileSync, mkdtempSync, rmSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { WhisperUnavailableError } from "../whisper/manager.js";
@@ -24,7 +24,7 @@ function dummyAudio(): { dir: string; input: string } {
return { dir, input };
}
describe("transcribe — whisper unavailable", () => {
describe("transcribe command", () => {
let dirs: string[] = [];
let priorExitCode: typeof process.exitCode;
@@ -66,4 +66,60 @@ describe("transcribe — whisper unavailable", () => {
expect(trackTranscribeUnavailable).toHaveBeenCalledWith({ optional: true });
expect(trackCommandFailure).not.toHaveBeenCalled();
});
it("imports an SRT and exports an SRT sidecar from transcript.json", async () => {
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
dirs.push(dir);
const input = join(dir, "sample.srt");
const sample = `1
00:00:01,000 --> 00:00:03,500
Write HTML.
2
00:00:03,500 --> 00:00:06,000
Render video. Built for agents.
`;
writeFileSync(input, sample);
await transcribeCmd.run!({ args: { input, dir, json: true } } as never);
const transcriptPath = join(dir, "transcript.json");
await transcribeCmd.run!({ args: { input: transcriptPath, to: "srt", json: true } } as never);
const outputPath = join(dir, "transcript.srt");
expect(readFileSync(outputPath, "utf-8")).toBe(sample);
const log = vi.mocked(console.log).mock.calls.at(-1)?.[0];
expect(typeof log).toBe("string");
if (typeof log !== "string") throw new Error("Expected JSON log output");
expect(JSON.parse(log)).toEqual({
ok: true,
format: "srt",
wordCount: 2,
outputPath,
});
});
it("--preserve-cues keeps single-word cues separate when exporting from JSON", async () => {
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
dirs.push(dir);
// Single-word cues have no internal whitespace, so the whitespace heuristic
// can't tell them from word-level whisper output. --preserve-cues forces 1:1.
const transcriptPath = join(dir, "transcript.json");
writeFileSync(
transcriptPath,
JSON.stringify([
{ text: "Yes", start: 0, end: 1 },
{ text: "No", start: 1, end: 2 },
]),
);
await transcribeCmd.run!({
args: { input: transcriptPath, to: "srt", "preserve-cues": true, json: true },
} as never);
const output = readFileSync(join(dir, "transcript.srt"), "utf-8");
expect(output).toBe(
"1\n00:00:00,000 --> 00:00:01,000\nYes\n\n2\n00:00:01,000 --> 00:00:02,000\nNo\n",
);
});
});
+97 -6
View File
@@ -2,6 +2,8 @@ import { defineCommand } from "citty";
import type { Example } from "./_examples.js";
import { existsSync, writeFileSync } from "node:fs";
type CaptionExportFormat = "srt" | "vtt";
export const examples: Example[] = [
["Transcribe an audio file", "hyperframes transcribe audio.mp3"],
["Transcribe a video file", "hyperframes transcribe video.mp4"],
@@ -9,6 +11,11 @@ export const examples: Example[] = [
["Set language to filter non-target speech", "hyperframes transcribe audio.mp3 --language en"],
["Import an existing SRT file", "hyperframes transcribe subtitles.srt"],
["Import an OpenAI Whisper JSON response", "hyperframes transcribe response.json"],
["Export captions to SRT", "hyperframes transcribe transcript.json --to srt"],
[
"Export single-word/CJK captions without re-grouping",
"hyperframes transcribe transcript.json --to vtt --preserve-cues",
],
];
import { resolve, join, extname, dirname } from "node:path";
import * as clack from "@clack/prompts";
@@ -49,6 +56,21 @@ export default defineCommand({
description: "Output result as JSON",
default: false,
},
to: {
type: "string",
description: "Export transcript sidecar format: srt or vtt",
},
output: {
type: "string",
alias: "o",
description: "Output path for exported SRT/VTT sidecar",
},
"preserve-cues": {
type: "boolean",
description:
"Keep each transcript entry as its own caption cue (skip word-level grouping). Use when exporting an already-cued transcript whose entries have no internal spaces, e.g. single-word or CJK captions.",
default: false,
},
optional: {
type: "boolean",
description:
@@ -73,6 +95,17 @@ export default defineCommand({
// ── Import mode: convert existing transcript ──────────────────────────
const isImport = ext === ".json" || ext === ".srt" || ext === ".vtt";
const to = parseExportFormat(args.to, args.json);
if (to) {
if (!isImport) {
failWith(
"--to can only export from transcript files (.json, .srt, .vtt). Run transcribe first.",
args.json,
);
}
return exportTranscript(inputPath, dir, to, args.output, args.json, args["preserve-cues"]);
}
if (isImport) {
return importTranscript(inputPath, dir, args.json);
@@ -88,20 +121,40 @@ export default defineCommand({
},
});
function failWith(message: string, json: boolean): never {
trackCommandFailure("transcribe", message);
if (json) {
console.log(JSON.stringify({ ok: false, error: message }));
} else {
console.error(c.error(message));
}
process.exit(1);
}
function parseExportFormat(
value: string | undefined,
json: boolean,
): CaptionExportFormat | undefined {
if (!value) return undefined;
const normalized = value.toLowerCase();
if (normalized === "srt" || normalized === "vtt") return normalized;
failWith(`Unsupported caption export format: ${value}. Use srt or vtt.`, json);
}
// ---------------------------------------------------------------------------
// Import existing transcript
// ---------------------------------------------------------------------------
function exitNoWords(json: boolean): never {
failWith("No words found in transcript.", json);
}
async function importTranscript(inputPath: string, dir: string, json: boolean): Promise<void> {
const { loadTranscript, patchCaptionHtml } = await import("../whisper/normalize.js");
const { words, format } = loadTranscript(inputPath);
if (words.length === 0) {
const message = "No words found in transcript.";
trackCommandFailure("transcribe", message);
console.error(c.error(message));
process.exit(1);
}
if (words.length === 0) exitNoWords(json);
const outPath = join(dir, "transcript.json");
writeFileSync(outPath, JSON.stringify(words, null, 2));
@@ -118,6 +171,44 @@ async function importTranscript(inputPath: string, dir: string, json: boolean):
}
}
// ---------------------------------------------------------------------------
// Export transcript sidecars
// ---------------------------------------------------------------------------
async function exportTranscript(
inputPath: string,
dir: string,
to: CaptionExportFormat,
output: string | undefined,
json: boolean,
preserveCues: boolean,
): Promise<void> {
const { loadTranscript, formatSrt, formatVtt } = await import("../whisper/normalize.js");
const { words, format } = loadTranscript(inputPath);
if (words.length === 0) exitNoWords(json);
// A .srt/.vtt source is already phrase-level; keep its cue boundaries 1:1.
// --preserve-cues forces the same for an already-cued transcript.json whose
// entries have no internal whitespace (single-word or CJK captions), which
// the automatic whitespace heuristic in wordsToCues can't detect.
const preGrouped = preserveCues || format === "srt" || format === "vtt" || undefined;
const outPath = resolve(output ?? join(dir, `transcript.${to}`));
const content =
to === "srt" ? formatSrt(words, { preGrouped }) : formatVtt(words, { preGrouped });
writeFileSync(outPath, content);
if (json) {
console.log(
JSON.stringify({ ok: true, format: to, wordCount: words.length, outputPath: outPath }),
);
} else {
console.log(
`${c.success("◇")} Exported ${c.accent(String(words.length))} words to ${c.accent(to.toUpperCase())}${c.accent(outPath)}`,
);
}
}
// ---------------------------------------------------------------------------
// Transcribe audio/video with whisper
// ---------------------------------------------------------------------------