fix(media-use): forgiving prompt matching + precise assets/ scan

Two defects in the resolve cascade that made cache and asset-reuse
misbehave in practice:

- Prompt matching was byte-exact and case-sensitive. findByPrompt and
  cacheGet compared provenance.prompt with ===, so "Calm piano" and
  "calm  piano" re-searched and re-downloaded instead of reusing the
  cached asset (same project and cross-project). Add normalizePrompt
  (trim + lowercase + collapse whitespace) and key both lookups on it;
  the raw prompt is still stored for audit.

- findExistingAsset matched with name.includes(intent) ||
  intent.includes(name), which silently returned the WRONG local file:
  intent "whoosh" grabbed a stray who.mp3, and a one-letter filename
  matched every intent. Require a shared word token (>= 3 chars, minus
  stopwords) so a false negative just falls through to a catalog search
  rather than shipping the wrong asset.

Adds lib/adopt.test.mjs and extends manifest.test.mjs. Full media-use
suite green; verified e2e against the live catalog (case-variant
cross-project resolve now reuses; whoosh no longer grabs who.mp3).
This commit is contained in:
Miguel Angel Simon Sierra
2026-07-07 15:21:41 -04:00
parent 2bd9bb6f69
commit b5383ded42
7 changed files with 179 additions and 13 deletions
+37 -4
View File
@@ -96,16 +96,49 @@ export function adoptExistingAssets(projectDir) {
return adopted;
}
// Common filler words that should never, on their own, make a filename match an
// intent (e.g. intent "the rocket" must not adopt "the-video.mp4").
const MATCH_STOPWORDS = new Set([
"the",
"and",
"for",
"with",
"from",
"this",
"that",
"your",
"our",
]);
// Split into lowercased word tokens of length >= 3, minus stopwords.
function matchTokens(text) {
return new Set(
String(text)
.toLowerCase()
.split(/[^a-z0-9]+/)
.filter((t) => t.length >= 3 && !MATCH_STOPWORDS.has(t)),
);
}
// Adopt a pre-existing assets/ file only when it shares a meaningful word with
// the intent. The old test — `name.includes(intent) || intent.includes(name)` —
// silently returned the WRONG file: "whoosh" grabbed a stray who.mp3, and a
// one-letter filename matched every intent. A false negative just falls through
// to a catalog search (safe); a false positive ships the wrong asset. So bias to
// precision: require a shared token, don't guess from substrings.
export function findExistingAsset(projectDir, intent, type) {
const assetsDir = join(projectDir, "assets");
if (!existsSync(assetsDir)) return null;
const lower = intent.toLowerCase();
const intentTokens = matchTokens(intent);
if (intentTokens.size === 0) return null;
for (const rel of walkDir(assetsDir)) {
const t = inferType(rel);
if (!t || (type && t !== type)) continue;
const name = basename(rel, extname(rel)).toLowerCase().replace(/[-_]/g, " ");
if (name.includes(lower) || lower.includes(name)) {
return { relativePath: `assets/${rel}`, type: t, name: basename(rel, extname(rel)) };
const stem = basename(rel, extname(rel));
for (const tok of matchTokens(stem)) {
if (intentTokens.has(tok)) {
return { relativePath: `assets/${rel}`, type: t, name: stem };
}
}
}
return null;