mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-05 17:30:50 +00:00
* feat(cli): add --lang and auto-infer phonemizer locale from voice prefix `hyperframes tts` was calling Kokoro's `model.create(text, voice=, speed=)` with no language argument, so Kokoro's default phonemizer (en-us) was applied regardless of the voice selected. Picking `ef_dora` or `jf_alpha` and feeding it Spanish or Japanese text produced English-phonemized output. Closes #349. - `manager.ts`: add `SUPPORTED_LANGS`, `inferLangFromVoiceId`, and `isSupportedLang`. Attach a `defaultLang` field to every bundled voice and expand the bundled list with `ef_dora`, `ff_siwis`, `jf_alpha`, `zf_xiaobei` so `--list` surfaces multilingual options. - `synthesize.ts`: accept optional `lang: SupportedLang` in `SynthesizeOptions`, forward it to the Python worker as `argv[7]`. The worker introspects `Kokoro.create`'s signature and only passes `lang=` when the installed kokoro-onnx version supports it. Returned metadata now includes `lang` and `langApplied` so callers can detect silent no-ops. Bump the cached script filename to `synth-v2.py` so existing installs pick up the new script automatically. - `commands/tts.ts`: add `--lang, -l` with validation against `SUPPORTED_LANGS`. Resolution order is explicit `--lang` > inferred from voice prefix > `en-us`. When explicit lang disagrees with the voice-implied lang (legitimate for stylized accents), emit a dim-level hint; suppress under `--json`. When kokoro-onnx silently ignores the kwarg, log that too. Update `--list` with a new "Lang code" column and add multilingual examples. - Tests: new `manager.test.ts` covering every supported prefix, the unknown-prefix fallback, case-insensitivity, `isSupportedLang` validation, and a regression guard that every bundled voice has a valid `defaultLang` matching its ID. - Docs: `docs/packages/cli.mdx` and `skills/hyperframes/references/tts.md` updated with the flag, examples, the espeak-ng dependency note for non-English phonemization, and the voice-prefix → lang table. Backward compatibility: - English voices (a*/b* prefixes) continue to phonemize as en-us / en-gb — no change. - Non-English voices now phonemize correctly by default (bug fix, not a regression). - Older kokoro-onnx versions that don't know the `lang` kwarg keep working via signature introspection; the CLI logs a dim note if `--lang` was requested but ignored. Verification: - `bun --cwd packages/cli test` — 128 tests pass (incl. 17 new). - `bunx oxlint` and `bunx oxfmt --check` clean on changed files. - `bun run build` succeeds. - `npx tsx packages/cli/src/cli.ts tts --help` / `--list` render cleanly; invalid `--lang` produces a clean error with the valid-codes list. * refactor(cli): simplify tts --lang implementation Post-review cleanup on #351. Net -21 lines. - Drop `defaultLang` field + `makeVoice()` helper from VoiceInfo — compute via `inferLangFromVoiceId(v.id)` at read time in listVoices. The only reader was the --list table; caching the derived value on every voice added a self-consistency invariant we had to test. - Drop redundant `lang` field from SynthesizeResult — caller already knows the requested lang since it passed it in; only `langApplied` carries information the caller can't derive. - Use `errorBox` for --lang validation to match the house style in render.ts (other validation errors already use errorBox). - Reuse existing `langList` module constant in the validation error instead of re-joining SUPPORTED_LANGS. - Inline `DEFAULT_LANG` — used once in inferLangFromVoiceId. - Trim WHAT-restating comments and the duplicate prefix-enumeration JSDoc on inferLangFromVoiceId (VOICE_PREFIX_LANG already carries per-row comments). - Clean up orphaned `synth*.py` files in ~/.cache/hyperframes/tts when writing the current versioned script, so repeated upgrades don't leak files. - Drop the `EN-US` case-sensitive-rejection test assertion — the CLI lowercases input before validation, so accepting mixed case is a feature, not a bug. Tests: 16/16 in `manager.test.ts`, 127/127 full CLI suite pass. Lint + format + typecheck clean.
149 lines
4.9 KiB
TypeScript
149 lines
4.9 KiB
TypeScript
import { existsSync, mkdirSync } from "node:fs";
|
|
import { homedir } from "node:os";
|
|
import { join } from "node:path";
|
|
import { downloadFile } from "../utils/download.js";
|
|
|
|
const CACHE_DIR = join(homedir(), ".cache", "hyperframes", "tts");
|
|
const MODELS_DIR = join(CACHE_DIR, "models");
|
|
const VOICES_DIR = join(CACHE_DIR, "voices");
|
|
|
|
const DEFAULT_MODEL = "kokoro-v1.0";
|
|
|
|
const MODEL_URLS: Record<string, string> = {
|
|
"kokoro-v1.0":
|
|
"https://github.com/thewh1teagle/kokoro-onnx/releases/download/model-files-v1.0/kokoro-v1.0.onnx",
|
|
};
|
|
|
|
const VOICES_URL =
|
|
"https://github.com/thewh1teagle/kokoro-onnx/releases/download/model-files-v1.0/voices-v1.0.bin";
|
|
|
|
// Locale codes accepted by Kokoro's phonemizer (misaki for English,
|
|
// espeak-ng for everything else). Kept as a readonly tuple so the union
|
|
// type below stays driven by this single source.
|
|
export const SUPPORTED_LANGS = [
|
|
"en-us",
|
|
"en-gb",
|
|
"es",
|
|
"fr-fr",
|
|
"hi",
|
|
"it",
|
|
"pt-br",
|
|
"ja",
|
|
"zh",
|
|
] as const;
|
|
|
|
export type SupportedLang = (typeof SUPPORTED_LANGS)[number];
|
|
|
|
// Kokoro voice IDs are `<lang><gender>_<name>` — the first letter is
|
|
// language, the second is gender. See https://github.com/hexgrad/kokoro.
|
|
const VOICE_PREFIX_LANG: Record<string, SupportedLang> = {
|
|
a: "en-us", // American English
|
|
b: "en-gb", // British English
|
|
e: "es", // Spanish
|
|
f: "fr-fr", // French
|
|
h: "hi", // Hindi
|
|
i: "it", // Italian
|
|
j: "ja", // Japanese
|
|
p: "pt-br", // Brazilian Portuguese
|
|
z: "zh", // Mandarin
|
|
};
|
|
|
|
/**
|
|
* Infer the phonemizer language from a Kokoro voice ID prefix.
|
|
* Unknown prefixes fall back to `en-us` — Kokoro's text frontend is
|
|
* English-trained, so that's the safe default.
|
|
*/
|
|
export function inferLangFromVoiceId(voiceId: string): SupportedLang {
|
|
const first = voiceId.charAt(0).toLowerCase();
|
|
return VOICE_PREFIX_LANG[first] ?? "en-us";
|
|
}
|
|
|
|
export function isSupportedLang(value: string): value is SupportedLang {
|
|
return (SUPPORTED_LANGS as readonly string[]).includes(value);
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Voices — Kokoro ships 54 voices across 8 languages. We expose a curated
|
|
// default set and allow users to specify any valid Kokoro voice ID.
|
|
// ---------------------------------------------------------------------------
|
|
|
|
export interface VoiceInfo {
|
|
id: string;
|
|
label: string;
|
|
language: string;
|
|
gender: "female" | "male";
|
|
}
|
|
|
|
export const BUNDLED_VOICES: VoiceInfo[] = [
|
|
{ id: "af_heart", label: "Heart", language: "en-US", gender: "female" },
|
|
{ id: "af_nova", label: "Nova", language: "en-US", gender: "female" },
|
|
{ id: "af_sky", label: "Sky", language: "en-US", gender: "female" },
|
|
{ id: "am_adam", label: "Adam", language: "en-US", gender: "male" },
|
|
{ id: "am_michael", label: "Michael", language: "en-US", gender: "male" },
|
|
{ id: "bf_emma", label: "Emma", language: "en-GB", gender: "female" },
|
|
{ id: "bf_isabella", label: "Isabella", language: "en-GB", gender: "female" },
|
|
{ id: "bm_george", label: "George", language: "en-GB", gender: "male" },
|
|
{ id: "ef_dora", label: "Dora", language: "es", gender: "female" },
|
|
{ id: "ff_siwis", label: "Siwis", language: "fr-FR", gender: "female" },
|
|
{ id: "jf_alpha", label: "Alpha", language: "ja", gender: "female" },
|
|
{ id: "zf_xiaobei", label: "Xiaobei", language: "zh", gender: "female" },
|
|
];
|
|
|
|
export const DEFAULT_VOICE = "af_heart";
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Public API
|
|
// ---------------------------------------------------------------------------
|
|
|
|
/**
|
|
* Ensure the Kokoro ONNX model is downloaded and cached.
|
|
* Returns the path to the .onnx model file.
|
|
*/
|
|
export async function ensureModel(
|
|
model: string = DEFAULT_MODEL,
|
|
options?: { onProgress?: (message: string) => void },
|
|
): Promise<string> {
|
|
const modelPath = join(MODELS_DIR, `${model}.onnx`);
|
|
if (existsSync(modelPath)) return modelPath;
|
|
|
|
const url = MODEL_URLS[model];
|
|
if (!url) {
|
|
throw new Error(
|
|
`Unknown TTS model: ${model}. Available: ${Object.keys(MODEL_URLS).join(", ")}`,
|
|
);
|
|
}
|
|
|
|
mkdirSync(MODELS_DIR, { recursive: true });
|
|
options?.onProgress?.(`Downloading TTS model ${model} (~311 MB)...`);
|
|
await downloadFile(url, modelPath);
|
|
|
|
if (!existsSync(modelPath)) {
|
|
throw new Error(`Model download failed: ${model}`);
|
|
}
|
|
|
|
return modelPath;
|
|
}
|
|
|
|
/**
|
|
* Ensure the Kokoro voices bundle is downloaded and cached.
|
|
* Returns the path to the voices .bin file.
|
|
*/
|
|
export async function ensureVoices(options?: {
|
|
onProgress?: (message: string) => void;
|
|
}): Promise<string> {
|
|
const voicesPath = join(VOICES_DIR, "voices-v1.0.bin");
|
|
if (existsSync(voicesPath)) return voicesPath;
|
|
|
|
mkdirSync(VOICES_DIR, { recursive: true });
|
|
options?.onProgress?.("Downloading voice data (~27 MB)...");
|
|
await downloadFile(VOICES_URL, voicesPath);
|
|
|
|
if (!existsSync(voicesPath)) {
|
|
throw new Error("Voice data download failed");
|
|
}
|
|
|
|
return voicesPath;
|
|
}
|
|
|
|
export { MODELS_DIR, VOICES_DIR, DEFAULT_MODEL };
|