Merge pull request #987 from heygen-com/feat/capture-font-extractor

feat(capture): identify hashed fonts via OpenType name table
This commit is contained in:
Ular Kimsanov
2026-05-21 10:09:25 -07:00
committed by GitHub
5 changed files with 578 additions and 9 deletions
+2
View File
@@ -30,6 +30,7 @@
"citty": "^0.2.1",
"compare-versions": "^6.1.1",
"esbuild": "^0.25.12",
"fontkit": "^2.0.4",
"giget": "^3.2.0",
"hono": "^4.0.0",
"onnxruntime-node": "^1.20.0",
@@ -47,6 +48,7 @@
"@hyperframes/producer": "workspace:*",
"@hyperframes/studio": "workspace:*",
"@types/adm-zip": "^0.5.7",
"@types/fontkit": "^2.0.9",
"@types/mime-types": "^3.0.1",
"@types/node": "^25.0.10",
"linkedom": "^0.18.12",
@@ -0,0 +1,176 @@
import { describe, expect, it } from "vitest";
import { mkdtempSync, rmSync, existsSync, readFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import {
canonicalizeFamily,
extractFontMetadata,
inferWeightFromSubfamily,
} from "./fontMetadataExtractor.js";
describe("inferWeightFromSubfamily", () => {
// The concatenated forms were always handled. The spaced and hyphenated
// forms were the bug Copilot flagged on PR #987 — "Extra Light" used to
// fall through to the 400 default before the whitespace-normalization fix.
describe("concatenated forms (already handled)", () => {
it.each([
["Thin", 100],
["ExtraLight", 200],
["UltraLight", 200],
["Light", 300],
["Regular", 400],
["Medium", 500],
["SemiBold", 600],
["DemiBold", 600],
["Bold", 700],
["ExtraBold", 800],
["UltraBold", 800],
["Black", 900],
["Heavy", 900],
])("%s → %d", (subfamily, expected) => {
expect(inferWeightFromSubfamily(subfamily)).toBe(expected);
});
});
describe("spaced forms (the bug fix)", () => {
it.each([
["Extra Light", 200],
["Ultra Light", 200],
["Semi Bold", 600],
["Demi Bold", 600],
["Extra Bold", 800],
["Ultra Bold", 800],
])("%s → %d", (subfamily, expected) => {
expect(inferWeightFromSubfamily(subfamily)).toBe(expected);
});
});
describe("hyphenated forms (the bug fix)", () => {
it.each([
["Extra-Light", 200],
["Semi-Bold", 600],
["Extra-Bold", 800],
])("%s → %d", (subfamily, expected) => {
expect(inferWeightFromSubfamily(subfamily)).toBe(expected);
});
});
describe("composite styles", () => {
it("Bold Italic still detects Bold", () => {
expect(inferWeightFromSubfamily("Bold Italic")).toBe(700);
});
it("Semi Bold Italic still detects SemiBold (priority over Bold)", () => {
expect(inferWeightFromSubfamily("Semi Bold Italic")).toBe(600);
});
it("ExtraBold Italic still detects ExtraBold (priority over Bold)", () => {
expect(inferWeightFromSubfamily("ExtraBold Italic")).toBe(800);
});
});
it("unknown subfamily falls back to 400 (Regular)", () => {
expect(inferWeightFromSubfamily("Headline")).toBe(400);
expect(inferWeightFromSubfamily("")).toBe(400);
expect(inferWeightFromSubfamily("Some Random Style")).toBe(400);
});
it("is case-insensitive", () => {
expect(inferWeightFromSubfamily("EXTRA LIGHT")).toBe(200);
expect(inferWeightFromSubfamily("extra light")).toBe(200);
expect(inferWeightFromSubfamily("ExTrA LiGhT")).toBe(200);
});
});
describe("canonicalizeFamily", () => {
it("returns family unchanged when no weight token is trailing", () => {
expect(canonicalizeFamily("Inter")).toEqual({
canonical: "Inter",
inferredWeight: null,
});
expect(canonicalizeFamily("Tiempos Headline")).toEqual({
canonical: "Tiempos Headline",
inferredWeight: null,
});
expect(canonicalizeFamily("Söhne Breit")).toEqual({
canonical: "Söhne Breit",
inferredWeight: null,
});
});
it("strips trailing weight tokens and surfaces the implied weight", () => {
expect(canonicalizeFamily("Inter Medium")).toEqual({
canonical: "Inter",
inferredWeight: 500,
});
expect(canonicalizeFamily("Inter Light")).toEqual({
canonical: "Inter",
inferredWeight: 300,
});
expect(canonicalizeFamily("Inter Bold")).toEqual({
canonical: "Inter",
inferredWeight: 700,
});
expect(canonicalizeFamily("Funnel Display Light")).toEqual({
canonical: "Funnel Display",
inferredWeight: 300,
});
});
it("preserves width modifiers before the weight token", () => {
expect(canonicalizeFamily("Inter Tight Medium")).toEqual({
canonical: "Inter Tight",
inferredWeight: 500,
});
});
it("emits 950 for ExtraBlack / UltraBlack (mirrors foundry intent)", () => {
expect(canonicalizeFamily("Inter ExtraBlack")).toEqual({
canonical: "Inter",
inferredWeight: 950,
});
});
it("returns empty input unchanged", () => {
expect(canonicalizeFamily("")).toEqual({
canonical: "",
inferredWeight: null,
});
});
});
describe("extractFontMetadata", () => {
// Light integration tests against the public surface — uses a real
// temp directory and verifies the manifest shape. Doesn't require
// fixture font binaries; the non-existent and empty-directory cases
// exercise the happy paths for the surrounding pipeline.
it("returns an empty manifest when the fonts directory doesn't exist", () => {
const tmp = mkdtempSync(join(tmpdir(), "hf-font-test-"));
try {
const outputPath = join(tmp, "manifest.json");
const manifest = extractFontMetadata(join(tmp, "does-not-exist"), outputPath);
expect(manifest.files).toEqual([]);
expect(manifest.families).toEqual([]);
expect(manifest.unidentified).toEqual([]);
expect(existsSync(outputPath)).toBe(true);
const written = JSON.parse(readFileSync(outputPath, "utf-8")) as typeof manifest;
expect(written.files).toEqual([]);
expect(written.meta.tool).toBe("fontkit");
expect(typeof written.meta.generatedAt).toBe("string");
} finally {
rmSync(tmp, { recursive: true, force: true });
}
});
it("writes a manifest with the documented meta shape", () => {
const tmp = mkdtempSync(join(tmpdir(), "hf-font-test-"));
try {
const outputPath = join(tmp, "manifest.json");
const manifest = extractFontMetadata(tmp, outputPath);
expect(manifest.meta.tool).toBe("fontkit"); // no version hardcoded — moves with the dep
// generatedAt is an ISO string
expect(manifest.meta.generatedAt).toMatch(/^\d{4}-\d{2}-\d{2}T/);
} finally {
rmSync(tmp, { recursive: true, force: true });
}
});
});
@@ -0,0 +1,339 @@
/**
* Extract font metadata from downloaded font files.
*
* Modern web frameworks (Next.js, Webpack) rename fonts with content hashes for
* cache-busting, leaving downloaded files like `19cfc7226ec3afaa-s.woff2` with
* no human-readable identification. The CSS @font-face mapping that originally
* tied each hash back to a family name is often lost during capture.
*
* Every OpenType / WOFF / WOFF2 file embeds a `name` table (part of the spec
* since 1996) containing the family, subfamily, full name, PostScript name,
* weight class, and variation axes. Subsetting and hashing do not strip it.
* This extractor uses `fontkit` to read the name table from each downloaded
* font and writes a manifest the rest of the pipeline can consult instead of
* guessing from filename patterns.
*
* Output: extracted/fonts-manifest.json with per-file metadata + per-family
* aggregation. See FontsManifest type for shape.
*/
import { readdirSync, readFileSync, writeFileSync, existsSync } from "node:fs";
import { join } from "node:path";
import * as fontkit from "fontkit";
import type { Font, FontCollection } from "fontkit";
function isFontCollection(value: Font | FontCollection): value is FontCollection {
return value.type === "TTC" || value.type === "DFont";
}
export interface FontFileMetadata {
/** Filename relative to capture/assets/fonts/ (e.g. "19cfc7226ec3afaa-s.woff2") */
file: string;
/**
* Canonical family name. Many static-weight font files package each weight as
* a separate "family" in nameID 1 (e.g. "Inter Medium" instead of "Inter").
* This field strips trailing weight tokens so multiple weights of the same
* typographic family aggregate cleanly. See rawFamily for the unmodified value.
*/
family: string;
/**
* Raw family name as extracted, before canonicalization. Source precedence:
* 1. OpenType `name` table (nameID 16 if present, else nameID 1)
* 2. Fallback: derived from the PostScript name (nameID 6) before the first
* `-` (e.g. PostScript "Inter-Regular" → "Inter")
* Empty string when both the name table and PostScript name are absent
* (i.e. when `identified` is false).
*/
rawFamily: string;
/** Subfamily / style name from nameID 17 or 2 (e.g. "Regular", "Bold Italic") */
subfamily: string;
/** PostScript name from nameID 6 (e.g. "Inter-Regular") */
postscript: string;
/**
* Weight value. Typically the OS/2 `usWeightClass` (100900) when present.
* Other values you may see:
* - `0`: returned when the file is `identified: false` (no name-table data
* to infer from); treat as unknown.
* - `950`: emitted by the family-name canonicalization when a foundry
* packaged "ExtraBlack" or "UltraBlack" as its own family. This is
* outside the 100-900 standard range but mirrors the foundry intent.
* For variable fonts, this is the file's default axis position — see
* `variationAxes` for the available `wght` range.
*/
weight: number;
/** "normal" or "italic" — derived from subfamily and OS/2 fsSelection */
style: "normal" | "italic";
/** If this is a variable font, the axes present (e.g. ["wght", "slnt"]). Empty for static fonts. */
variationAxes: string[];
/** Whether identification came from the binary name table (the trustworthy source). */
identified: boolean;
}
export interface FontFamilySummary {
/** Family name */
family: string;
/** Distinct weights captured (from OS/2 weight class — for variable fonts shows the default) */
weights: number[];
/** Whether any file in this family is a variable font */
variable: boolean;
/** Number of files in this family (typically subsets of the same weight) */
fileCount: number;
/** Files in this family — useful for picking the @font-face src */
files: string[];
}
export interface FontsManifest {
/** Per-file metadata, one entry per downloaded font */
files: FontFileMetadata[];
/** Aggregated per-family summary — most useful for DESIGN.md authoring */
families: FontFamilySummary[];
/** Files where identification failed entirely. Should be empty for typical captures. */
unidentified: string[];
/** Generated-at timestamp + tool version for debugging */
meta: { generatedAt: string; tool: string };
}
/**
* Read all font files in fontsDir, extract metadata via fontkit, and write
* the manifest to outputPath. Returns the manifest in case callers want to log it.
*
* Failures are non-fatal: if a single font's name table is missing or corrupt,
* the file is added to `unidentified` and the rest continue. If the fonts
* directory doesn't exist, returns an empty manifest without throwing.
*/
export function extractFontMetadata(fontsDir: string, outputPath: string): FontsManifest {
const files: FontFileMetadata[] = [];
const unidentified: string[] = [];
if (existsSync(fontsDir)) {
const fontFiles = readdirSync(fontsDir).filter((f) => /\.(woff2?|ttf|otf)$/i.test(f));
for (const filename of fontFiles) {
const fullPath = join(fontsDir, filename);
const meta = readSingleFont(fullPath, filename);
if (meta.identified) {
files.push(meta);
} else {
files.push(meta);
unidentified.push(filename);
}
}
}
const families = aggregateFamilies(files);
const manifest: FontsManifest = {
files,
families,
unidentified,
meta: {
generatedAt: new Date().toISOString(),
// Record just the tool name; the version moves with the dep and would
// otherwise drift from a hardcoded string on every fontkit bump.
tool: "fontkit",
},
};
writeFileSync(outputPath, JSON.stringify(manifest, null, 2), "utf-8");
return manifest;
}
// fallow-ignore-next-line complexity
function readSingleFont(fullPath: string, filename: string): FontFileMetadata {
const empty: FontFileMetadata = {
file: filename,
family: "",
rawFamily: "",
subfamily: "",
postscript: "",
weight: 0,
style: "normal",
variationAxes: [],
identified: false,
};
try {
const buf = readFileSync(fullPath);
// fontkit.create returns Font | FontCollection. For TTC/DFont collections,
// take the first font inside; otherwise the value is already a single Font.
const created: Font | FontCollection = fontkit.create(buf);
const font: Font | undefined = isFontCollection(created) ? created.fonts[0] : created;
if (!font) return empty;
const rawFamily = (font.familyName || "").trim();
const subfamily = (font.subfamilyName || "").trim();
const postscript = (font.postscriptName || "").trim();
const fsSelection = font["OS/2"]?.fsSelection;
const italicBit = Boolean(fsSelection?.italic || fsSelection?.oblique);
const style: "normal" | "italic" =
italicBit || /italic|oblique/i.test(subfamily) ? "italic" : "normal";
const variationAxes = font.variationAxes ? Object.keys(font.variationAxes) : [];
if (!rawFamily && !postscript) return empty; // name table empty — cannot identify
const familyForCanonicalization = rawFamily || deriveFamilyFromPostscript(postscript);
const { canonical, inferredWeight } = canonicalizeFamily(familyForCanonicalization);
const weight =
font["OS/2"]?.usWeightClass ?? inferredWeight ?? inferWeightFromSubfamily(subfamily);
return {
file: filename,
family: canonical || familyForCanonicalization,
rawFamily: familyForCanonicalization,
subfamily,
postscript,
weight,
style,
variationAxes,
identified: true,
};
} catch {
return empty;
}
}
/** Aggregate per-file entries into per-family summaries — most useful shape for DESIGN.md. */
// fallow-ignore-next-line complexity
function aggregateFamilies(files: FontFileMetadata[]): FontFamilySummary[] {
const byFamily = new Map<string, FontFamilySummary>();
for (const f of files) {
if (!f.family) continue;
let entry = byFamily.get(f.family);
if (!entry) {
entry = { family: f.family, weights: [], variable: false, fileCount: 0, files: [] };
byFamily.set(f.family, entry);
}
entry.fileCount++;
entry.files.push(f.file);
if (f.variationAxes.length > 0) entry.variable = true;
if (f.weight && !entry.weights.includes(f.weight)) entry.weights.push(f.weight);
}
for (const entry of byFamily.values()) {
entry.weights.sort((a, b) => a - b);
entry.files.sort();
}
return Array.from(byFamily.values()).sort((a, b) => a.family.localeCompare(b.family));
}
/**
* PostScript names follow the convention `Family-Style`. When the family name
* record (nameID 1) is missing but PostScript is present, recover the family
* portion as a best-effort fallback.
*/
function deriveFamilyFromPostscript(postscript: string): string {
if (!postscript) return "";
const dashIdx = postscript.indexOf("-");
return (dashIdx > 0 ? postscript.slice(0, dashIdx) : postscript).trim();
}
/**
* Fallback when OS/2 table is missing — guess weight from "Bold", "Light", etc.
*
* Normalizes spaces and hyphens out of the subfamily before matching so that
* fonts using spaced names ("Extra Light", "Semi Bold") or hyphenated names
* ("Extra-Light", "Semi-Bold") resolve to the same weight as the concatenated
* forms ("ExtraLight", "SemiBold"). Without this, a font subfamily of
* "Extra Light" would fall through every concat check and end at the 400
* default, misreporting a 200-weight font as 400.
*
* Exported for unit testing.
*/
// fallow-ignore-next-line complexity
export function inferWeightFromSubfamily(subfamily: string): number {
const s = subfamily.toLowerCase().replace(/[\s-]+/g, "");
if (s.includes("thin")) return 100;
if (s.includes("extralight") || s.includes("ultralight")) return 200;
if (s.includes("light")) return 300;
if (s.includes("medium")) return 500;
if (s.includes("semibold") || s.includes("demibold")) return 600;
if (s.includes("extrabold") || s.includes("ultrabold")) return 800;
if (s.includes("black") || s.includes("heavy")) return 900;
if (s.includes("bold")) return 700;
return 400;
}
/**
* Map of trailing weight tokens found in family names (e.g. "Inter Medium" →
* "Inter") to their numeric OS/2 weight equivalent. Used to canonicalize family
* names when a foundry packaged each weight as a separate "family" instead of
* setting nameID 16 / 17 (Preferred Family / Subfamily).
*
* Conservative: only strips well-known English weight tokens. Width modifiers
* like "Tight", "Condensed", "Extended" are intentionally NOT stripped — they
* denote separate typographic families, not weight variants. Localized weight
* tokens (German "Fett", "Extrafett"; French "Maigre"; etc.) and abbreviations
* ("ExtBd", "ExtBlk") are not stripped either — the resulting family stays
* separate, which is an honest representation of what's in the file.
*/
const WEIGHT_TOKEN_TO_VALUE: Record<string, number> = {
Thin: 100,
Hairline: 100,
ExtraLight: 200,
UltraLight: 200,
Light: 300,
Book: 400,
Regular: 400,
Normal: 400,
Medium: 500,
SemiBold: 600,
DemiBold: 600,
Bold: 700,
ExtraBold: 800,
UltraBold: 800,
Black: 900,
Heavy: 900,
ExtraBlack: 950,
UltraBlack: 950,
};
const WEIGHT_TOKEN_RE = new RegExp(`\\s+(${Object.keys(WEIGHT_TOKEN_TO_VALUE).join("|")})$`, "i");
/**
* Strip a trailing weight token from a family name and return both the
* canonicalized form and the weight value the stripped token implied.
*
* Examples:
* "Inter Medium" → { canonical: "Inter", inferredWeight: 500 }
* "Inter Tight Medium" → { canonical: "Inter Tight", inferredWeight: 500 }
* "Funnel Display Light" → { canonical: "Funnel Display", inferredWeight: 300 }
* "Tiempos Headline" → { canonical: "Tiempos Headline", inferredWeight: null }
* "Söhne Breit Extrafett" → { canonical: "Söhne Breit Extrafett", inferredWeight: null }
*
* Trailing "Italic"/"Oblique" is stripped before weight detection so families
* like "Inter Italic" or "Inter Medium Italic" canonicalize correctly. The
* italic flag is recovered separately from the OS/2 fsSelection bit, so no
* information is lost.
*/
// Exported for unit testing.
// fallow-ignore-next-line complexity
export function canonicalizeFamily(family: string): {
canonical: string;
inferredWeight: number | null;
} {
if (!family) return { canonical: family, inferredWeight: null };
let result = family.trim();
// Strip trailing "Italic" or "Oblique" first — handled by the style field.
result = result.replace(/\s+(Italic|Oblique)$/i, "").trim();
// Normalize compound weight tokens written with a space ("Semi Bold" → "SemiBold")
// so the single-token matcher below catches them. Anchored to end-of-string to
// avoid touching family names that legitimately contain these words mid-string.
result = result.replace(
/\s+(Semi|Extra|Ultra|Demi)\s+(Bold|Black|Light)$/i,
(_, prefix: string, suffix: string) => ` ${capitalize(prefix)}${capitalize(suffix)}`,
);
// Strip trailing weight token if any.
const match = result.match(WEIGHT_TOKEN_RE);
if (match && match[1]) {
// Look up the canonical (case-sensitive) key for the matched token.
const matchedKey = Object.keys(WEIGHT_TOKEN_TO_VALUE).find(
(k) => k.toLowerCase() === match[1]!.toLowerCase(),
);
const inferredWeight = matchedKey ? WEIGHT_TOKEN_TO_VALUE[matchedKey]! : null;
result = result.slice(0, result.length - match[0].length).trim();
return { canonical: result, inferredWeight };
}
return { canonical: result, inferredWeight: null };
}
function capitalize(s: string): string {
return s.length === 0 ? s : s[0]!.toUpperCase() + s.slice(1).toLowerCase();
}
+28
View File
@@ -16,6 +16,7 @@ import { extractHtml } from "./htmlExtractor.js";
// captureScreenshots removed — full-page screenshot replaces per-section shots
import { extractTokens } from "./tokenExtractor.js";
import { downloadAssets, downloadAndRewriteFonts } from "./assetDownloader.js";
import { extractFontMetadata } from "./fontMetadataExtractor.js";
// briefGenerator.ts, visual-style, capture-summary removed — DESIGN.md replaces them
import {
setupAnimationCapture,
@@ -419,6 +420,33 @@ export async function captureWebsite(
// Download fonts and rewrite URLs to local paths
extracted.headHtml = await downloadAndRewriteFonts(extracted.headHtml, outputDir);
// Identify each downloaded font by reading its OpenType name table.
// Modern frameworks hash font filenames; this manifest tells the
// downstream pipeline (DESIGN.md authoring, beat sub-agents) which file
// belongs to which family without guessing from filename patterns.
try {
const fontsManifest = extractFontMetadata(
join(outputDir, "assets", "fonts"),
join(outputDir, "extracted", "fonts-manifest.json"),
);
if (fontsManifest.families.length > 0) {
const summary = fontsManifest.families
.map((f) => `${f.family}${f.variable ? " (variable)" : ""} × ${f.fileCount}`)
.join(", ");
console.log(`Font metadata extracted: ${summary}`);
if (fontsManifest.unidentified.length > 0) {
console.warn(
` ${fontsManifest.unidentified.length} font file(s) could not be identified — DESIGN.md should flag these explicitly.`,
);
}
}
} catch (err) {
console.warn(
"Font metadata extraction failed (non-fatal):",
err instanceof Error ? err.message : err,
);
}
// Save animation catalog — lean version for the agent (not 745 raw CSS declarations)
if (animationCatalog) {
// Extract just what's useful: counts, named animations, a few representative keyframed entries