mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-11 06:30:03 +00:00
fix(media-use): forgiving prompt matching + precise assets/ scan
Two defects in the resolve cascade that made cache and asset-reuse misbehave in practice: - Prompt matching was byte-exact and case-sensitive. findByPrompt and cacheGet compared provenance.prompt with ===, so "Calm piano" and "calm piano" re-searched and re-downloaded instead of reusing the cached asset (same project and cross-project). Add normalizePrompt (trim + lowercase + collapse whitespace) and key both lookups on it; the raw prompt is still stored for audit. - findExistingAsset matched with name.includes(intent) || intent.includes(name), which silently returned the WRONG local file: intent "whoosh" grabbed a stray who.mp3, and a one-letter filename matched every intent. Require a shared word token (>= 3 chars, minus stopwords) so a false negative just falls through to a catalog search rather than shipping the wrong asset. Adds lib/adopt.test.mjs and extends manifest.test.mjs. Full media-use suite green; verified e2e against the live catalog (case-variant cross-project resolve now reuses; whoosh no longer grabs who.mp3).
This commit is contained in:
@@ -96,16 +96,49 @@ export function adoptExistingAssets(projectDir) {
|
||||
return adopted;
|
||||
}
|
||||
|
||||
// Common filler words that should never, on their own, make a filename match an
|
||||
// intent (e.g. intent "the rocket" must not adopt "the-video.mp4").
|
||||
const MATCH_STOPWORDS = new Set([
|
||||
"the",
|
||||
"and",
|
||||
"for",
|
||||
"with",
|
||||
"from",
|
||||
"this",
|
||||
"that",
|
||||
"your",
|
||||
"our",
|
||||
]);
|
||||
|
||||
// Split into lowercased word tokens of length >= 3, minus stopwords.
|
||||
function matchTokens(text) {
|
||||
return new Set(
|
||||
String(text)
|
||||
.toLowerCase()
|
||||
.split(/[^a-z0-9]+/)
|
||||
.filter((t) => t.length >= 3 && !MATCH_STOPWORDS.has(t)),
|
||||
);
|
||||
}
|
||||
|
||||
// Adopt a pre-existing assets/ file only when it shares a meaningful word with
|
||||
// the intent. The old test — `name.includes(intent) || intent.includes(name)` —
|
||||
// silently returned the WRONG file: "whoosh" grabbed a stray who.mp3, and a
|
||||
// one-letter filename matched every intent. A false negative just falls through
|
||||
// to a catalog search (safe); a false positive ships the wrong asset. So bias to
|
||||
// precision: require a shared token, don't guess from substrings.
|
||||
export function findExistingAsset(projectDir, intent, type) {
|
||||
const assetsDir = join(projectDir, "assets");
|
||||
if (!existsSync(assetsDir)) return null;
|
||||
const lower = intent.toLowerCase();
|
||||
const intentTokens = matchTokens(intent);
|
||||
if (intentTokens.size === 0) return null;
|
||||
for (const rel of walkDir(assetsDir)) {
|
||||
const t = inferType(rel);
|
||||
if (!t || (type && t !== type)) continue;
|
||||
const name = basename(rel, extname(rel)).toLowerCase().replace(/[-_]/g, " ");
|
||||
if (name.includes(lower) || lower.includes(name)) {
|
||||
return { relativePath: `assets/${rel}`, type: t, name: basename(rel, extname(rel)) };
|
||||
const stem = basename(rel, extname(rel));
|
||||
for (const tok of matchTokens(stem)) {
|
||||
if (intentTokens.has(tok)) {
|
||||
return { relativePath: `assets/${rel}`, type: t, name: stem };
|
||||
}
|
||||
}
|
||||
}
|
||||
return null;
|
||||
|
||||
Reference in New Issue
Block a user