feat(cli): export .srt/.vtt caption sidecars from a transcript (#1704)

Add formatSrt/formatVtt/wordsToCues to normalize.ts (the inverse of the
existing parseSrt/parseVtt) and a 'hyperframes transcribe <transcript> --to
srt|vtt' export mode. Word-level whisper transcripts group into cues on
sentence boundaries with maxChars/maxGap guards; imported phrase-level cues
pass through unchanged. Default transcribe behavior is unchanged and no new
dependencies are added.

Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>
This commit is contained in:
Matt Van Horn
2026-06-24 19:03:16 -04:00
committed by GitHub
co-authored by Matt Van Horn
parent 97db811a2f
commit 821bf2921a
5 changed files with 428 additions and 11 deletions
+58 -2
View File
@@ -1,5 +1,5 @@
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
import { writeFileSync, mkdtempSync, rmSync } from "node:fs";
import { writeFileSync, readFileSync, mkdtempSync, rmSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { WhisperUnavailableError } from "../whisper/manager.js";
@@ -24,7 +24,7 @@ function dummyAudio(): { dir: string; input: string } {
return { dir, input };
}
describe("transcribe — whisper unavailable", () => {
describe("transcribe command", () => {
let dirs: string[] = [];
let priorExitCode: typeof process.exitCode;
@@ -66,4 +66,60 @@ describe("transcribe — whisper unavailable", () => {
expect(trackTranscribeUnavailable).toHaveBeenCalledWith({ optional: true });
expect(trackCommandFailure).not.toHaveBeenCalled();
});
it("imports an SRT and exports an SRT sidecar from transcript.json", async () => {
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
dirs.push(dir);
const input = join(dir, "sample.srt");
const sample = `1
00:00:01,000 --> 00:00:03,500
Write HTML.
2
00:00:03,500 --> 00:00:06,000
Render video. Built for agents.
`;
writeFileSync(input, sample);
await transcribeCmd.run!({ args: { input, dir, json: true } } as never);
const transcriptPath = join(dir, "transcript.json");
await transcribeCmd.run!({ args: { input: transcriptPath, to: "srt", json: true } } as never);
const outputPath = join(dir, "transcript.srt");
expect(readFileSync(outputPath, "utf-8")).toBe(sample);
const log = vi.mocked(console.log).mock.calls.at(-1)?.[0];
expect(typeof log).toBe("string");
if (typeof log !== "string") throw new Error("Expected JSON log output");
expect(JSON.parse(log)).toEqual({
ok: true,
format: "srt",
wordCount: 2,
outputPath,
});
});
it("--preserve-cues keeps single-word cues separate when exporting from JSON", async () => {
const dir = mkdtempSync(join(tmpdir(), "hf-transcribe-test-"));
dirs.push(dir);
// Single-word cues have no internal whitespace, so the whitespace heuristic
// can't tell them from word-level whisper output. --preserve-cues forces 1:1.
const transcriptPath = join(dir, "transcript.json");
writeFileSync(
transcriptPath,
JSON.stringify([
{ text: "Yes", start: 0, end: 1 },
{ text: "No", start: 1, end: 2 },
]),
);
await transcribeCmd.run!({
args: { input: transcriptPath, to: "srt", "preserve-cues": true, json: true },
} as never);
const output = readFileSync(join(dir, "transcript.srt"), "utf-8");
expect(output).toBe(
"1\n00:00:00,000 --> 00:00:01,000\nYes\n\n2\n00:00:01,000 --> 00:00:02,000\nNo\n",
);
});
});
+97 -6
View File
@@ -2,6 +2,8 @@ import { defineCommand } from "citty";
import type { Example } from "./_examples.js";
import { existsSync, writeFileSync } from "node:fs";
type CaptionExportFormat = "srt" | "vtt";
export const examples: Example[] = [
["Transcribe an audio file", "hyperframes transcribe audio.mp3"],
["Transcribe a video file", "hyperframes transcribe video.mp4"],
@@ -9,6 +11,11 @@ export const examples: Example[] = [
["Set language to filter non-target speech", "hyperframes transcribe audio.mp3 --language en"],
["Import an existing SRT file", "hyperframes transcribe subtitles.srt"],
["Import an OpenAI Whisper JSON response", "hyperframes transcribe response.json"],
["Export captions to SRT", "hyperframes transcribe transcript.json --to srt"],
[
"Export single-word/CJK captions without re-grouping",
"hyperframes transcribe transcript.json --to vtt --preserve-cues",
],
];
import { resolve, join, extname, dirname } from "node:path";
import * as clack from "@clack/prompts";
@@ -49,6 +56,21 @@ export default defineCommand({
description: "Output result as JSON",
default: false,
},
to: {
type: "string",
description: "Export transcript sidecar format: srt or vtt",
},
output: {
type: "string",
alias: "o",
description: "Output path for exported SRT/VTT sidecar",
},
"preserve-cues": {
type: "boolean",
description:
"Keep each transcript entry as its own caption cue (skip word-level grouping). Use when exporting an already-cued transcript whose entries have no internal spaces, e.g. single-word or CJK captions.",
default: false,
},
optional: {
type: "boolean",
description:
@@ -73,6 +95,17 @@ export default defineCommand({
// ── Import mode: convert existing transcript ──────────────────────────
const isImport = ext === ".json" || ext === ".srt" || ext === ".vtt";
const to = parseExportFormat(args.to, args.json);
if (to) {
if (!isImport) {
failWith(
"--to can only export from transcript files (.json, .srt, .vtt). Run transcribe first.",
args.json,
);
}
return exportTranscript(inputPath, dir, to, args.output, args.json, args["preserve-cues"]);
}
if (isImport) {
return importTranscript(inputPath, dir, args.json);
@@ -88,20 +121,40 @@ export default defineCommand({
},
});
function failWith(message: string, json: boolean): never {
trackCommandFailure("transcribe", message);
if (json) {
console.log(JSON.stringify({ ok: false, error: message }));
} else {
console.error(c.error(message));
}
process.exit(1);
}
function parseExportFormat(
value: string | undefined,
json: boolean,
): CaptionExportFormat | undefined {
if (!value) return undefined;
const normalized = value.toLowerCase();
if (normalized === "srt" || normalized === "vtt") return normalized;
failWith(`Unsupported caption export format: ${value}. Use srt or vtt.`, json);
}
// ---------------------------------------------------------------------------
// Import existing transcript
// ---------------------------------------------------------------------------
function exitNoWords(json: boolean): never {
failWith("No words found in transcript.", json);
}
async function importTranscript(inputPath: string, dir: string, json: boolean): Promise<void> {
const { loadTranscript, patchCaptionHtml } = await import("../whisper/normalize.js");
const { words, format } = loadTranscript(inputPath);
if (words.length === 0) {
const message = "No words found in transcript.";
trackCommandFailure("transcribe", message);
console.error(c.error(message));
process.exit(1);
}
if (words.length === 0) exitNoWords(json);
const outPath = join(dir, "transcript.json");
writeFileSync(outPath, JSON.stringify(words, null, 2));
@@ -118,6 +171,44 @@ async function importTranscript(inputPath: string, dir: string, json: boolean):
}
}
// ---------------------------------------------------------------------------
// Export transcript sidecars
// ---------------------------------------------------------------------------
async function exportTranscript(
inputPath: string,
dir: string,
to: CaptionExportFormat,
output: string | undefined,
json: boolean,
preserveCues: boolean,
): Promise<void> {
const { loadTranscript, formatSrt, formatVtt } = await import("../whisper/normalize.js");
const { words, format } = loadTranscript(inputPath);
if (words.length === 0) exitNoWords(json);
// A .srt/.vtt source is already phrase-level; keep its cue boundaries 1:1.
// --preserve-cues forces the same for an already-cued transcript.json whose
// entries have no internal whitespace (single-word or CJK captions), which
// the automatic whitespace heuristic in wordsToCues can't detect.
const preGrouped = preserveCues || format === "srt" || format === "vtt" || undefined;
const outPath = resolve(output ?? join(dir, `transcript.${to}`));
const content =
to === "srt" ? formatSrt(words, { preGrouped }) : formatVtt(words, { preGrouped });
writeFileSync(outPath, content);
if (json) {
console.log(
JSON.stringify({ ok: true, format: to, wordCount: words.length, outputPath: outPath }),
);
} else {
console.log(
`${c.success("◇")} Exported ${c.accent(String(words.length))} words to ${c.accent(to.toUpperCase())}${c.accent(outPath)}`,
);
}
}
// ---------------------------------------------------------------------------
// Transcribe audio/video with whisper
// ---------------------------------------------------------------------------
+98 -1
View File
@@ -2,7 +2,15 @@ import { describe, it, expect, afterEach } from "vitest";
import { writeFileSync, readFileSync, mkdirSync, rmSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { loadTranscript, detectFormat, patchCaptionHtml, stripBeforeOnset } from "./normalize.js";
import {
loadTranscript,
detectFormat,
patchCaptionHtml,
stripBeforeOnset,
formatSrt,
formatVtt,
wordsToCues,
} from "./normalize.js";
import { detectSpeechOnset } from "./transcribe.js";
function tmpFile(name: string, content: string): string {
@@ -213,6 +221,95 @@ Short format
});
});
describe("caption formatting", () => {
it("round-trips SRT cues through normalized words", () => {
const srt = `1
00:00:01,000 --> 00:00:03,500
Write HTML.
2
00:00:03,500 --> 00:00:06,000
Render video. Built for agents.
`;
const path = tmpFile("captions.srt", srt);
const { words } = loadTranscript(path);
const output = formatSrt(words);
expect(output).toBe(srt);
const reparsed = loadTranscript(tmpFile("roundtrip.srt", output));
expect(reparsed.words).toEqual(words);
});
it("round-trips VTT cues through normalized words", () => {
const vtt = `WEBVTT
00:00:01.000 --> 00:00:03.500
Write HTML.
00:00:03.500 --> 00:00:06.000
Render video. Built for agents.
`;
const path = tmpFile("captions.vtt", vtt);
const { words } = loadTranscript(path);
const output = formatVtt(words);
expect(output).toBe(vtt);
const reparsed = loadTranscript(tmpFile("roundtrip.vtt", output));
expect(reparsed.words).toEqual(words);
});
it("groups word-level transcript entries into readable cues", () => {
const cues = wordsToCues(
[
{ text: "Write", start: 0, end: 0.2 },
{ text: "HTML.", start: 0.2, end: 0.5 },
{ text: "Render", start: 0.7, end: 0.9 },
{ text: "video", start: 0.9, end: 1.1 },
{ text: "for", start: 1.1, end: 1.2 },
{ text: "agents.", start: 1.2, end: 1.6 },
{ text: "Fresh", start: 2.5, end: 2.8 },
{ text: "tracks.", start: 3.9, end: 4.1 },
],
{ maxChars: 18, maxGap: 0.8 },
);
expect(cues).toEqual([
{ text: "Write HTML.", start: 0, end: 0.5 },
{ text: "Render video for", start: 0.7, end: 1.2 },
{ text: "agents.", start: 1.2, end: 1.6 },
{ text: "Fresh", start: 2.5, end: 2.8 },
{ text: "tracks.", start: 3.9, end: 4.1 },
]);
});
it("joins CJK word-level tokens without inserting spaces", () => {
const cues = wordsToCues([
{ text: "你", start: 0, end: 0.3 },
{ text: "好", start: 0.3, end: 0.6 },
{ text: "世界", start: 0.6, end: 1.0 },
]);
expect(cues).toEqual([{ text: "你好世界", start: 0, end: 1 }]);
});
it("preserves single-word cue boundaries when preGrouped", () => {
// Phrase-level cues without internal whitespace (one-word or CJK captions)
// must not merge — auto-detection can't see them, so the caller forces it.
const cues = wordsToCues(
[
{ text: "Yes", start: 0, end: 1 },
{ text: "No", start: 1, end: 2 },
],
{ preGrouped: true },
);
expect(cues).toEqual([
{ text: "Yes", start: 0, end: 1 },
{ text: "No", start: 1, end: 2 },
]);
});
});
describe("whisper-cpp contraction merging", () => {
it("merges didn + 't into didn't", () => {
const path = tmpFile(
+165
View File
@@ -11,6 +11,20 @@ export interface Word {
end: number;
}
export interface Cue {
text: string;
start: number;
end: number;
}
export interface WordsToCuesOptions {
maxChars?: number;
maxGap?: number;
/** Treat each entry as a finished cue (skip word-level grouping). Defaults to
* auto-detection: true when any entry contains internal whitespace. */
preGrouped?: boolean;
}
// ---------------------------------------------------------------------------
// Format detection + parsing
// ---------------------------------------------------------------------------
@@ -262,10 +276,161 @@ function parseVttTimestamp(ts: string): number {
return 0;
}
/** Format SRT timestamp: seconds → 00:01:23,456 */
function formatSrtTimestamp(seconds: number): string {
const { hours, minutes, wholeSeconds, milliseconds } = timestampParts(seconds);
return `${pad2(hours)}:${pad2(minutes)}:${pad2(wholeSeconds)},${pad3(milliseconds)}`;
}
/** Format VTT timestamp: seconds → 00:01:23.456 */
function formatVttTimestamp(seconds: number): string {
const { hours, minutes, wholeSeconds, milliseconds } = timestampParts(seconds);
return `${pad2(hours)}:${pad2(minutes)}:${pad2(wholeSeconds)}.${pad3(milliseconds)}`;
}
function round3(n: number): number {
return Math.round(n * 1000) / 1000;
}
function timestampParts(seconds: number): {
hours: number;
minutes: number;
wholeSeconds: number;
milliseconds: number;
} {
const safeSeconds = Number.isFinite(seconds) ? seconds : 0;
const totalMs = Math.max(0, Math.round(safeSeconds * 1000));
const milliseconds = totalMs % 1000;
const totalSeconds = (totalMs - milliseconds) / 1000;
const wholeSeconds = totalSeconds % 60;
const totalMinutes = (totalSeconds - wholeSeconds) / 60;
const minutes = totalMinutes % 60;
const hours = (totalMinutes - minutes) / 60;
return { hours, minutes, wholeSeconds, milliseconds };
}
function pad2(n: number): string {
return n.toString().padStart(2, "0");
}
function pad3(n: number): string {
return n.toString().padStart(3, "0");
}
function endsSentence(text: string): boolean {
return /[.!?][)"'\]}]*$/.test(text);
}
function pushCue(cues: Cue[], cue: Cue | undefined): void {
if (!cue) return;
const text = cue.text.trim();
if (!text) return;
cues.push({ text, start: round3(cue.start), end: round3(cue.end) });
}
/** Whether `word` should start a new cue rather than extend `current`. */
function breaksCue(
current: Cue,
word: Word,
text: string,
maxChars: number,
maxGap: number,
): boolean {
const nextLength = current.text.length + 1 + text.length;
const gap = word.start - current.end;
return nextLength > maxChars || gap > maxGap;
}
/** Map each entry to its own cue (used when entries are already phrase-level). */
function entriesToCues(words: Word[]): Cue[] {
const cues: Cue[] = [];
for (const word of words) {
pushCue(cues, { text: word.text, start: word.start, end: word.end });
}
return cues;
}
// Han + Hiragana + Katakana + CJK symbols/fullwidth. These scripts are written
// without spaces between tokens, so whisper's per-token output must be joined
// without a separator. Hangul (Korean) is intentionally excluded — it does use
// inter-word spaces.
const CJK_CHAR = /[ -〿぀-ヿ㐀-䶿一-鿿豈-﫿＀-￯]/;
/** Join two adjacent tokens, omitting the space across a CJK boundary. */
function joinTokens(left: string, right: string): string {
const a = left.at(-1) ?? "";
const b = right[0] ?? "";
const sep = CJK_CHAR.test(a) || CJK_CHAR.test(b) ? "" : " ";
return `${left}${sep}${right}`;
}
export function wordsToCues(words: Word[], opts: WordsToCuesOptions = {}): Cue[] {
// Phrase-level transcripts (imported .srt/.vtt cues) must keep their existing
// cue boundaries — re-grouping would merge distinct captions and lose timing.
// The caller can force this via `preGrouped`; otherwise infer it from the data
// (any entry containing internal whitespace is a multi-word phrase, so the
// whole transcript is phrase-level rather than word-level whisper output).
const preGrouped = opts.preGrouped ?? words.some((w) => /\s/.test(w.text.trim()));
if (preGrouped) return entriesToCues(words);
const maxChars = opts.maxChars ?? 42;
const maxGap = opts.maxGap ?? 0.8;
const cues: Cue[] = [];
let current: Cue | undefined;
const flush = (): void => {
pushCue(cues, current);
current = undefined;
};
for (const word of words) {
const text = word.text.trim();
if (!text) continue;
if (current && !breaksCue(current, word, text, maxChars, maxGap)) {
current.text = joinTokens(current.text, text);
current.end = word.end;
} else {
flush();
current = { text, start: word.start, end: word.end };
}
if (endsSentence(text)) flush();
}
flush();
return cues;
}
export function formatSrt(words: Word[], opts?: WordsToCuesOptions): string {
const cues = wordsToCues(words, opts);
if (cues.length === 0) return "";
return (
cues
.map(
(cue, i) =>
`${i + 1}\n${formatSrtTimestamp(cue.start)} --> ${formatSrtTimestamp(cue.end)}\n${cue.text}`,
)
.join("\n\n") + "\n"
);
}
export function formatVtt(words: Word[], opts?: WordsToCuesOptions): string {
const cues = wordsToCues(words, opts);
if (cues.length === 0) return "WEBVTT\n\n";
return (
"WEBVTT\n\n" +
cues
.map(
(cue) => `${formatVttTimestamp(cue.start)} --> ${formatVttTimestamp(cue.end)}\n${cue.text}`,
)
.join("\n\n") +
"\n"
);
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------