mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-07 01:56:04 +00:00
* feat(media-use): resolve official brand logos via a four-tier cascade Third-party brand logos (the meeting's 'credibility signals lost' gap) had no acquisition path: capture only grabs the product's own site assets, and HeyGen asset search returns generic look-alike icons for brand queries (0/3 in testing — an X-in-a-circle for LinkedIn). Workers could only fake a mark or drop it. New resolve type 'logo', four tiers verified by a 54-brand stress test (100% cascade hit across dev tools / big tech / non-tech / CN brands): - svgl — official full-color vector SVGs + wordmark variants (40/54 first-hits); search is substring-based, so entities pass through alias normalization (nextjs → 'next.js', aws → 'amazon web services') - simple-icons (pinned CDN build) — official monochrome glyphs; catches the long tail (nike, visa, toyota, wechat, bytedance) - github org avatar — known-org map only; a brand name is not a GitHub login, guessing risks same-named personal accounts - domain favicon (DuckDuckGo ip3) — small-raster last resort; sub-500B responses are DDG's placeholder and rejected; frozen with a low_res provenance flag (chip-size use only) logo joins the icon/image equivalence group (typesMatch) and the images/ subdir, so entity cache hits interop with figma-imported marks. A total miss falls through resolve's normal failure path — no special casing. HeyGen search stays the icon provider; it is deliberately absent from the logo cascade. Docs: media-use gap/types/providers tables + example; the five workflow banners now cover logos (catalog claim kept for media, 'from their official sources' added for logos); product-launch story-design and motion-graphics logo-reveal point at the new type; catalog surfaces (CLAUDE.md / README / docs) updated in lockstep. Verified: 19 unit tests + coverage row green; live smoke across all four tiers (linkedin→svgl, nike→simple-icons, heygen→github.avatar, amazon→favicon) plus a fabricated brand exiting 1 on the default miss path. oxlint + oxfmt clean. * test(media-use): sanction the four logo providers in the registry allowlist svgl / simple-icons / github.avatar / favicon.ddg join the sanctioned list — the logo cascade added in the previous commit. Full lib suite 95/95 green. * test(media-use): gate the logo cascade behavior in CI + single-fetch favicon tier Review follow-ups (miga-heygen, jrusso1020 on #2061): - Eight mocked-network tests pin what the manual 54-brand stress test only asserted: descriptor shape, alias retry (svgl non-array payload → next query, simple-icons 404 → next slug), network-error → null fallthrough, the sub-500B placeholder rejection, github's no-guessing (zero fetches for unmapped entities), and the real cascade order landing tier by tier under a mocked network. - faviconSearch now hands its verified bytes over as a local file, so the freeze step copies instead of re-downloading — one round-trip, and the size check is authoritative over what gets frozen. - The header's hit counts are labeled as a stress-test snapshot, not a live invariant. Full lib suite 103/103; live smoke re-verified (amazon → favicon.ddg, frozen .ico).
190 lines
5.7 KiB
JavaScript
190 lines
5.7 KiB
JavaScript
import {
|
|
readFileSync,
|
|
appendFileSync,
|
|
mkdirSync,
|
|
existsSync,
|
|
readdirSync,
|
|
openSync,
|
|
closeSync,
|
|
writeFileSync,
|
|
rmSync,
|
|
statSync,
|
|
} from "node:fs";
|
|
import { join } from "node:path";
|
|
|
|
const MANIFEST_FILE = "manifest.jsonl";
|
|
const INDEX_FILE = "index.md";
|
|
|
|
const TYPE_DIRS = {
|
|
bgm: "audio/bgm",
|
|
sfx: "audio/sfx",
|
|
voice: "audio/voice",
|
|
image: "images",
|
|
icon: "images",
|
|
logo: "images",
|
|
brand: "images",
|
|
video: "video",
|
|
};
|
|
|
|
export function mediaDir(projectDir) {
|
|
return join(projectDir, ".media");
|
|
}
|
|
|
|
export function manifestPath(projectDir) {
|
|
return join(mediaDir(projectDir), MANIFEST_FILE);
|
|
}
|
|
|
|
export function indexPath(projectDir) {
|
|
return join(mediaDir(projectDir), INDEX_FILE);
|
|
}
|
|
|
|
export function typeSubdir(type) {
|
|
const sub = TYPE_DIRS[type];
|
|
if (!sub) throw new Error(`unknown media type: ${type}`);
|
|
return sub;
|
|
}
|
|
|
|
export function typeDirPath(projectDir, type) {
|
|
return join(mediaDir(projectDir), typeSubdir(type));
|
|
}
|
|
|
|
export function readManifest(projectDir) {
|
|
const p = manifestPath(projectDir);
|
|
if (!existsSync(p)) return [];
|
|
const raw = readFileSync(p, "utf8");
|
|
const records = [];
|
|
for (const line of raw.split(/\r?\n/)) {
|
|
const trimmed = line.trim();
|
|
if (!trimmed) continue;
|
|
try {
|
|
records.push(JSON.parse(trimmed));
|
|
} catch {
|
|
// ponytail: skip malformed lines, don't crash
|
|
}
|
|
}
|
|
return records;
|
|
}
|
|
|
|
export function appendRecord(projectDir, record) {
|
|
const dir = mediaDir(projectDir);
|
|
mkdirSync(dir, { recursive: true });
|
|
const typeDir = typeDirPath(projectDir, record.type);
|
|
mkdirSync(typeDir, { recursive: true });
|
|
|
|
const p = manifestPath(projectDir);
|
|
const line = JSON.stringify(record) + "\n";
|
|
appendFileSync(p, line);
|
|
}
|
|
|
|
// Match prompts forgivingly. Agents rarely re-emit a byte-identical intent, so
|
|
// keying cache lookups on exact equality meant "Calm piano" and "calm piano"
|
|
// re-searched and re-downloaded. Normalize (trim, lowercase, collapse internal
|
|
// whitespace) on both sides; the raw prompt is still stored for audit.
|
|
export function normalizePrompt(prompt) {
|
|
return String(prompt ?? "")
|
|
.trim()
|
|
.toLowerCase()
|
|
.replace(/\s+/g, " ");
|
|
}
|
|
|
|
export function findByPrompt(projectDir, prompt, type) {
|
|
const key = normalizePrompt(prompt);
|
|
if (!key) return null;
|
|
const records = readManifest(projectDir);
|
|
return (
|
|
records.find(
|
|
(r) => normalizePrompt(r.provenance?.prompt) === key && (type == null || r.type === type),
|
|
) || null
|
|
);
|
|
}
|
|
|
|
export function findByEntity(projectDir, entity) {
|
|
const lower = entity.toLowerCase();
|
|
const records = readManifest(projectDir);
|
|
return records.find((r) => r.entity && r.entity.toLowerCase() === lower) || null;
|
|
}
|
|
|
|
export function nextId(projectDir, type) {
|
|
const records = readManifest(projectDir);
|
|
const prefix = type;
|
|
let max = 0;
|
|
for (const r of records) {
|
|
if (r.type !== type) continue;
|
|
const m = r.id?.match(new RegExp(`^${prefix}_(\\d+)$`));
|
|
if (m) max = Math.max(max, parseInt(m[1], 10));
|
|
}
|
|
return `${prefix}_${String(max + 1).padStart(3, "0")}`;
|
|
}
|
|
|
|
// Sync sleep (no busy-spin) for the allocation lock retry.
|
|
function sleepMs(ms) {
|
|
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
}
|
|
|
|
// Coarse per-project lock so concurrent resolves don't race on id allocation.
|
|
// ponytail: one lock file with a 15s stale-steal (a crashed holder can't wedge
|
|
// the project); fine for agent-scale concurrency — revisit if throughput needs
|
|
// finer locking. Date.now() is available here (a normal Node CLI, not a
|
|
// workflow DSL), so mtime-based staleness is safe.
|
|
const LOCK_STALE_MS = 15000;
|
|
const LOCK_TIMEOUT_MS = 20000;
|
|
|
|
function withLock(dir, fn) {
|
|
const lock = join(dir, ".lock");
|
|
const start = Date.now();
|
|
for (;;) {
|
|
try {
|
|
closeSync(openSync(lock, "wx")); // O_EXCL: atomic acquire
|
|
break;
|
|
} catch (err) {
|
|
if (err.code !== "EEXIST") throw err;
|
|
try {
|
|
if (Date.now() - statSync(lock).mtimeMs > LOCK_STALE_MS) {
|
|
rmSync(lock, { force: true }); // steal a stale lock from a dead holder
|
|
continue;
|
|
}
|
|
} catch {
|
|
continue; // lock vanished between check and stat — retry the acquire
|
|
}
|
|
if (Date.now() - start > LOCK_TIMEOUT_MS) {
|
|
throw new Error("media-use: timed out acquiring .media/.lock");
|
|
}
|
|
sleepMs(25);
|
|
}
|
|
}
|
|
try {
|
|
return fn();
|
|
} finally {
|
|
rmSync(lock, { force: true });
|
|
}
|
|
}
|
|
|
|
// Atomically allocate the next free id for `type` AND reserve its file, so a
|
|
// slow download/copy between allocation and appendRecord can't let a concurrent
|
|
// caller grab the same id (the MU-23 clobber). Under the lock we take the max id
|
|
// across BOTH the manifest and any already-reserved files in the type dir, then
|
|
// O_EXCL-create an empty placeholder at the target path; freeze/copy overwrites
|
|
// it. Returns { id, localPath }.
|
|
export function allocateId(projectDir, type, ext) {
|
|
mkdirSync(mediaDir(projectDir), { recursive: true });
|
|
const typeDir = typeDirPath(projectDir, type);
|
|
mkdirSync(typeDir, { recursive: true });
|
|
return withLock(mediaDir(projectDir), () => {
|
|
const re = new RegExp(`^${type}_(\\d+)`);
|
|
let max = 0;
|
|
for (const r of readManifest(projectDir)) {
|
|
if (r.type !== type) continue;
|
|
const m = r.id?.match(re);
|
|
if (m) max = Math.max(max, parseInt(m[1], 10));
|
|
}
|
|
for (const f of readdirSync(typeDir)) {
|
|
const m = f.match(re);
|
|
if (m) max = Math.max(max, parseInt(m[1], 10)); // skip ids reserved but not yet appended
|
|
}
|
|
const id = `${type}_${String(max + 1).padStart(3, "0")}`;
|
|
const localPath = `.media/${typeSubdir(type)}/${id}${ext}`;
|
|
writeFileSync(join(projectDir, localPath), "", { flag: "wx" }); // durable reservation
|
|
return { id, localPath };
|
|
});
|
|
}
|