mirror of
https://github.com/heygen-com/hyperframes.git
synced 2026-09-05 00:56:23 +00:00
Windows sizes Python's stdio and text-mode file IO to the ANSI code page
(cp1252), not UTF-8. Every skill Python script relied on that default:
* analyze-beatgrid.py --print writes the glyphs cp1252 has no slot for
(delta, arrow), so the brief died with UnicodeEncodeError on every Windows
run — the reported crash;
* its audiomap write_text() pairs ensure_ascii=False with the default file
encoding, so a non-ASCII payload is unwritable there too;
* lint_source.py read_text() raises UnicodeDecodeError before any rule runs
when a Remotion source carries an em dash or a curly quote;
* gen-stroke-path.py reads an SVG font whose glyph keys ARE literal
characters, so a mis-decoded key stops matching the requested text.
Stdio is reconfigured to UTF-8 at import and every text-mode IO call names its
encoding. `errors` is carried across the reconfigure: it resets to "strict",
and CPython gives stderr "backslashreplace" on purpose so the diagnostic path
can never itself raise.
extract-audio-data.py also decoded ffmpeg's stderr strictly while reporting a
failure, which would bury the very error being reported on a Windows ffmpeg.
skills/python-encoding.test.mjs guards the class: it fails if any skill Python
script drops the stdio block or omits encoding= on a text-mode IO call. The
mode is read as a whole comma-delimited argument of mode characters only, so a
payload key like {"bpm": 120} cannot spell the check away.
Verified with a cp1252 stdio stream installed before module load, matching how
Windows starts the interpreter: pre-fix UnicodeEncodeError, post-fix both
glyphs present in the UTF-8 bytes. Not run on real Windows hardware.
543 lines
24 KiB
Python
543 lines
24 KiB
Python
#!/usr/bin/env python3
|
||
"""Beat-grid + drum/event analysis engine for music-to-video.
|
||
|
||
Turns a BGM track directly into a deterministic `audiomap.json` — the music skeleton
|
||
that the Director and Builder hang visuals on. It merges:
|
||
|
||
- a reliable tempo + beat grid + downbeat (librosa beat tracker),
|
||
- metrical position per event (strong / weak / syncopated / off-grid) over a 16th-note bar grid,
|
||
- drum-element classification (kick / snare / hihat / perc) via band-split,
|
||
- special audio events (riser / glitch / crash-impact / hard-stop silence),
|
||
- an energy narrative (audio-driven energy phases + builds + key moments; the Music
|
||
Reader names the sections — no fixed Intro/Build/Drop/Outro template),
|
||
- a phrase layer + per-section density budgets for visual planning.
|
||
|
||
Output is the canonical `audiomap.json` documented in the skill.
|
||
|
||
Usage:
|
||
python3 analyze-beatgrid.py track.mp3 -o audiomap.json
|
||
python3 analyze-beatgrid.py track.mp3 --print # also print a readable brief
|
||
|
||
Deps: ffmpeg/ffprobe on PATH + librosa, numpy, soundfile (band-split heuristics,
|
||
no learned models / no madmom).
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import subprocess
|
||
import sys
|
||
import tempfile
|
||
from pathlib import Path
|
||
|
||
import librosa
|
||
import numpy as np
|
||
import soundfile as sf
|
||
|
||
# Windows sizes stdio to the ANSI code page (cp1252), which cannot encode the glyphs
|
||
# the brief prints (Δ, →) — every `--print` run died with UnicodeEncodeError. These
|
||
# scripts emit UTF-8 on every platform; say so instead of trading the glyphs away.
|
||
# Carry `errors` across: reconfigure() resets it to "strict", and CPython deliberately
|
||
# gives stderr "backslashreplace" so the diagnostic path can never itself raise.
|
||
for _stream in (sys.stdout, sys.stderr):
|
||
if hasattr(_stream, "reconfigure"):
|
||
_stream.reconfigure(encoding="utf-8", errors=_stream.errors)
|
||
|
||
SR = 22050
|
||
HOP = 512 # ~23 ms frames
|
||
AUDIOMAP_VERSION = 2
|
||
|
||
|
||
# ── decode ────────────────────────────────────────────────────────────────
|
||
def load_audio(path: str) -> tuple[np.ndarray, int, float]:
|
||
"""Decode any ffmpeg-readable file to mono float32 @ SR via a temp wav."""
|
||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
||
wav = tmp.name
|
||
subprocess.run(
|
||
["ffmpeg", "-y", "-i", path, "-ac", "1", "-ar", str(SR), wav],
|
||
capture_output=True, check=True,
|
||
)
|
||
y, sr = sf.read(wav, dtype="float32")
|
||
Path(wav).unlink(missing_ok=True)
|
||
if y.ndim > 1:
|
||
y = y.mean(axis=1)
|
||
return y, sr, len(y) / sr
|
||
|
||
|
||
# ── tempo + beat grid + downbeat phase ──────────────────────────────────────
|
||
def beat_grid(y: np.ndarray, sr: int) -> dict:
|
||
tempo, beat_frames = librosa.beat.beat_track(y=y, sr=sr, hop_length=HOP, units="frames")
|
||
beats = librosa.frames_to_time(beat_frames, sr=sr, hop_length=HOP)
|
||
return {"bpm": float(np.atleast_1d(tempo)[0]), "beats": beats, "beat_frames": beat_frames}
|
||
|
||
|
||
def _norm_flux(band: np.ndarray) -> np.ndarray:
|
||
"""Positive first-difference (onset flux) of a band-energy curve, normalized."""
|
||
flux = np.maximum(0.0, np.diff(np.sqrt(band), prepend=band[:1]))
|
||
return flux / (flux.max() + 1e-9)
|
||
|
||
|
||
def band_energy_curves(y: np.ndarray, sr: int) -> dict:
|
||
"""Per-frame band energy + per-band normalized onset flux (for drum typing)."""
|
||
S = np.abs(librosa.stft(y, hop_length=HOP)) ** 2
|
||
freqs = librosa.fft_frequencies(sr=sr)
|
||
# Tight drum bands: kick fundamental, snare body, hihat sizzle. Narrow bands
|
||
# keep one drum's transient from leaking into another's flux on a full mix.
|
||
low = (freqs < 150) # kick
|
||
mid = (freqs >= 150) & (freqs < 900) # snare body
|
||
high = (freqs >= 5000) # hihat / cymbal
|
||
e_low, e_mid, e_high = S[low].sum(axis=0), S[mid].sum(axis=0), S[high].sum(axis=0)
|
||
return {
|
||
"S": S,
|
||
"low": e_low, "mid": e_mid, "high": e_high,
|
||
"total": S.sum(axis=0) + 1e-9,
|
||
"flux_low": _norm_flux(e_low), "flux_mid": _norm_flux(e_mid), "flux_high": _norm_flux(e_high),
|
||
"flatness": librosa.feature.spectral_flatness(S=np.sqrt(S))[0],
|
||
"centroid": librosa.feature.spectral_centroid(S=np.sqrt(S), sr=sr)[0],
|
||
"n": S.shape[1],
|
||
}
|
||
|
||
|
||
def downbeat_phase(beat_frames: np.ndarray, bc: dict, beats_per_bar: int = 4) -> int:
|
||
"""Pick the bar phase whose beats carry the most KICK (low-band) energy."""
|
||
kick = bc["low"] / bc["total"]
|
||
best_p, best_score = 0, -1.0
|
||
for p in range(beats_per_bar):
|
||
idx = [bf for i, bf in enumerate(beat_frames) if (i - p) % beats_per_bar == 0]
|
||
idx = [min(f, bc["n"] - 1) for f in idx]
|
||
score = float(np.sum([kick[f] for f in idx])) if idx else 0.0
|
||
if score > best_score:
|
||
best_p, best_score = p, score
|
||
return best_p
|
||
|
||
|
||
# ── metrical position: strong / weak / syncopated / off-grid ─────────────────
|
||
# 16-step bar grid (4 beats x 4 sixteenths). Strength by metrical weight.
|
||
GRID_CLASS = {0: "strong", 8: "strong", 4: "weak", 12: "weak",
|
||
2: "weak", 6: "weak", 10: "weak", 14: "weak"} # else (odd 16ths) -> syncopated
|
||
|
||
|
||
def classify_metric(t: float, beats: np.ndarray, phase: int, bpb: int = 4) -> tuple:
|
||
"""Return (grid_class, bar, beat_in_bar, step16) for a time t."""
|
||
if len(beats) < 2:
|
||
return "off-grid", -1, -1, -1
|
||
i = int(np.searchsorted(beats, t) - 1)
|
||
i = max(0, min(i, len(beats) - 2))
|
||
beat_dur = beats[i + 1] - beats[i]
|
||
frac = (t - beats[i]) / max(beat_dur, 1e-6) # 0..1 within the beat
|
||
sixteenth = int(round(frac * 4)) % 4 # nearest 16th in beat
|
||
carry = 1 if round(frac * 4) >= 4 else 0
|
||
beat_idx = i + carry
|
||
beat_in_bar = (beat_idx - phase) % bpb # 0..3
|
||
bar = (beat_idx - phase) // bpb
|
||
step16 = beat_in_bar * 4 + sixteenth # 0..15
|
||
# distance to the nearest 16th line (in seconds) → off-grid test
|
||
nearest = beats[i] + (round(frac * 4) / 4) * beat_dur
|
||
if abs(t - nearest) > 0.5 * (beat_dur / 4):
|
||
return "off-grid", bar, beat_in_bar + 1, step16
|
||
return GRID_CLASS.get(step16, "syncopated"), bar, beat_in_bar + 1, step16
|
||
|
||
|
||
# ── drum classification (band-split heuristic) ──────────────────────────────
|
||
def frame_at(t: float, sr: int, n: int) -> int:
|
||
return min(int(round(t * sr / HOP)), n - 1)
|
||
|
||
|
||
def classify_drum(t: float, bc: dict, sr: int) -> tuple:
|
||
"""(drum_type, energy_norm, feel): which band's onset TRANSIENT dominates.
|
||
|
||
Uses per-band normalized flux (relative transient strength), the standard
|
||
way to separate kick (low) / snare (mid+noise) / hihat (high). Falls back to
|
||
glitch for noisy non-harmonic bursts and perc when no band clearly leads.
|
||
"""
|
||
f = frame_at(t, sr, bc["n"])
|
||
win = slice(max(0, f - 1), min(bc["n"], f + 2))
|
||
fl = float(bc["flux_low"][win].max())
|
||
fm = float(bc["flux_mid"][win].max())
|
||
fh = float(bc["flux_high"][win].max())
|
||
lo = float(bc["low"][win].mean()); md = float(bc["mid"][win].mean())
|
||
hi = float(bc["high"][win].mean()); tot = float(bc["total"][win].mean())
|
||
flat = float(bc["flatness"][win].mean())
|
||
lr, mr, hr = lo / tot, md / tot, hi / tot
|
||
|
||
fluxes = {"kick": fl, "snare": fm, "hihat": fh}
|
||
lead = max(fluxes, key=fluxes.get)
|
||
lead_val = fluxes[lead]
|
||
if lead_val < 0.06: # no real transient → texture/perc
|
||
drum = "glitch" if flat > 0.30 else "perc"
|
||
elif lead == "snare" and flat > 0.30 and mr < 0.30:
|
||
drum = "glitch" # mid-band but noisy & thin → scratch/glitch
|
||
else:
|
||
drum = lead
|
||
# feel (frequency character)
|
||
has_bot, has_top, has_mid = lr > 0.30, hr > 0.20, mr > 0.30
|
||
feel = ("full" if has_bot and has_top and has_mid else
|
||
"heavy" if has_bot and not has_top else
|
||
"bright" if has_top and not has_bot else
|
||
"intimate" if has_mid else "sparse")
|
||
return drum, tot, feel
|
||
|
||
|
||
# ── energy structure (RMS @1s) + sections + key moments + builds ────────────
|
||
def energy_structure(y: np.ndarray, sr: int, dur: float, first_onset: float = 0.0) -> dict:
|
||
rms = librosa.feature.rms(y=y, hop_length=sr)[0] # ~1s frames
|
||
rms = rms / (rms.max() + 1e-9)
|
||
norms = rms.tolist()
|
||
|
||
def lvl(n):
|
||
return "VOID" if n < 0.2 else "LOW" if n < 0.4 else "MEDIUM" if n < 0.65 else "HIGH"
|
||
|
||
phases, cur, cs = [], None, 0
|
||
for i, n in enumerate(norms):
|
||
l = lvl(n)
|
||
if l != cur:
|
||
if cur:
|
||
phases.append({"s": cs, "e": i, "lvl": cur})
|
||
cur, cs = l, i
|
||
if cur:
|
||
phases.append({"s": cs, "e": len(norms), "lvl": cur})
|
||
|
||
moments = []
|
||
for i in range(1, len(norms)):
|
||
d = norms[i] - norms[i - 1]
|
||
if abs(d) > 0.12:
|
||
moments.append({"t": i, "kind": "DROP" if d < 0 else "SURGE", "delta": round(d, 2)})
|
||
moments.sort(key=lambda m: abs(m["delta"]), reverse=True)
|
||
|
||
# hard stop: a HIGH→low cliff (a sudden stop) in the back third
|
||
hard_stops = [m for m in moments if m["kind"] == "DROP" and m["t"] > dur * 0.6 and m["delta"] < -0.25]
|
||
|
||
# NO forced Intro/Build/Drop/Outro template. The energy phases (audio-driven runs of
|
||
# one energy level, variable count) are the raw structural blocks. The Music Reader
|
||
# (LLM) decides the actual sections — count, boundaries, and free-form names — from
|
||
# these phases + key_moments + rolls + hard_stops + phrases. Sections are
|
||
# interpretation; only the timing they snap to is fact.
|
||
phases_sec = []
|
||
for p in phases:
|
||
seg = norms[p["s"]:max(p["s"] + 1, p["e"])]
|
||
phases_sec.append({
|
||
"start": float(p["s"]),
|
||
"end": float(min(p["e"], round(dur, 1))),
|
||
"level": p["lvl"],
|
||
"energy": round(float(np.mean(seg)) if seg else 0.0, 2),
|
||
})
|
||
|
||
return {"norms": [round(n, 2) for n in norms], "phases": phases_sec,
|
||
"moments": moments[:8], "hard_stops": hard_stops}
|
||
|
||
|
||
# ── rolls / fills (localized rapid-onset runs) ───────────────────────────────
|
||
# A roll is where choreography should switch from discrete hits to a continuous /
|
||
# cascading visual (per-letter cascade, stagger). Derived straight from the onset
|
||
# stream — runs never overlap, so no dedup is needed (unlike a band-energy detector).
|
||
ROLL_MIN_HITS = 4
|
||
ROLL_CONT = 0.55 # × beat_dur: gap up to ~half a beat still keeps a run alive
|
||
ROLL_ACCEPT = 0.42 # × beat_dur: mean spacing denser than an 8th note counts
|
||
ROLL_DEDUP = 0.08 # seconds: merge onsets closer than a 32nd (double-trigger)
|
||
|
||
|
||
def detect_rolls(events: list, beat_dur: float) -> list:
|
||
"""Runs of >=4 onsets whose MEAN spacing is denser than an 8th note. Tuned so a
|
||
full hihat/snare roll is captured as ONE span (not fragmented down to its tail),
|
||
while a sparse groove stays out — validated against the golden 7.5-9.5s roll.
|
||
The linear scan means runs never overlap (no LEGACY-style double-counting)."""
|
||
cont = beat_dur * ROLL_CONT # max gap that keeps a run alive
|
||
accept = beat_dur * ROLL_ACCEPT # max MEAN gap for a run to count
|
||
# collapse onset double-triggers (two onsets < a 32nd apart = one hit) so a
|
||
# held/sparse passage can't masquerade as a roll on a duplicated transient.
|
||
ev = []
|
||
for e in events:
|
||
if ev and e["t"] - ev[-1]["t"] <= ROLL_DEDUP:
|
||
if e.get("energy", 0) > ev[-1].get("energy", 0):
|
||
ev[-1] = e
|
||
continue
|
||
ev.append(e)
|
||
times = [e["t"] for e in ev]
|
||
n = len(times)
|
||
rolls, i = [], 0
|
||
while i < n - 1:
|
||
j = i
|
||
while j + 1 < n and (times[j + 1] - times[j]) <= cont:
|
||
j += 1
|
||
if j - i + 1 >= ROLL_MIN_HITS:
|
||
gaps = [times[k + 1] - times[k] for k in range(i, j)]
|
||
if sum(gaps) / len(gaps) <= accept:
|
||
t0, t1 = times[i], times[j]
|
||
half = len(gaps) // 2
|
||
accel = (half >= 1 and
|
||
sum(gaps[half:]) / (len(gaps) - half) <
|
||
sum(gaps[:half]) / half * 0.85)
|
||
dcount: dict[str, int] = {}
|
||
for e in ev[i:j + 1]:
|
||
dcount[e["drum"]] = dcount.get(e["drum"], 0) + 1
|
||
rolls.append({
|
||
"start": round(t0, 3), "end": round(t1, 3),
|
||
"dur_sec": round(t1 - t0, 3),
|
||
"hits": j - i + 1,
|
||
"rate_per_min": round((j - i) / max(t1 - t0, 1e-6) * 60),
|
||
"kind": "accel-roll" if accel else ("sustained-fill" if t1 - t0 > 1.2 else "fill"),
|
||
"drum": max(dcount, key=dcount.get),
|
||
})
|
||
i = j + 1
|
||
return rolls
|
||
|
||
|
||
# ── per-section spectral character (sustained "feel") ────────────────────────
|
||
# Coarse, reliable bands for how a SECTION sounds (not the noisy per-second dump).
|
||
# Distinct from a per-event `feel`, which is the transient color of a single hit.
|
||
FEEL_BANDS = [("sub", 0, 60), ("bass", 60, 250), ("low_mid", 250, 800),
|
||
("mid", 800, 2500), ("presence", 2500, 6000), ("air", 6000, 1e9)]
|
||
|
||
|
||
def annotate_section_feel(bc: dict, sr: int, sections: list) -> None:
|
||
"""Attach {character, bands} to each energy phase from its sustained band balance."""
|
||
S, n = bc["S"], bc["n"]
|
||
freqs = librosa.fft_frequencies(sr=sr)
|
||
fps = sr / HOP
|
||
masks = [(name, (freqs >= lo) & (freqs < hi)) for name, lo, hi in FEEL_BANDS]
|
||
for s in sections:
|
||
f0 = int(s["start"] * fps)
|
||
f1 = min(max(f0 + 1, int(s["end"] * fps)), n)
|
||
seg = S[:, f0:f1]
|
||
if seg.shape[1] == 0 or s.get("energy", 0) < 0.15:
|
||
s["feel"] = {"character": "sparse", "bands": []}
|
||
continue
|
||
en = {name: float(seg[m].sum()) for name, m in masks}
|
||
tot = sum(en.values()) + 1e-9
|
||
ratios = {name: en[name] / tot for name in en}
|
||
present = sorted([bn for bn, r in ratios.items() if r > 0.12],
|
||
key=lambda bn: -ratios[bn])
|
||
lo = ratios["sub"] + ratios["bass"]
|
||
hi = ratios["presence"] + ratios["air"]
|
||
mid = ratios["low_mid"] + ratios["mid"]
|
||
char = ("heavy" if lo > 0.5 else "bright" if hi > 0.45 else
|
||
"full" if lo > 0.25 and hi > 0.25 else
|
||
"warm" if mid > 0.5 else "sparse")
|
||
s["feel"] = {"character": char, "bands": present}
|
||
|
||
|
||
# ── audiomap enrichment: phrase layer + section density budgets ─────────────
|
||
def round3(n: float) -> float:
|
||
return round(float(n), 3)
|
||
|
||
|
||
def derive_phrases(downbeats: list[float], phrase_bars: int, duration_sec: float) -> list:
|
||
"""Group downbeats into phrase spans of `phrase_bars` bars."""
|
||
phrases = []
|
||
if not downbeats:
|
||
return phrases
|
||
index = 0
|
||
for i in range(0, len(downbeats), phrase_bars):
|
||
start = downbeats[i]
|
||
next_idx = i + phrase_bars
|
||
end = downbeats[next_idx] if next_idx < len(downbeats) else duration_sec
|
||
phrases.append({
|
||
"index": index,
|
||
"start": round3(start),
|
||
"end": round3(end),
|
||
"bars": min(phrase_bars, len(downbeats) - i),
|
||
})
|
||
index += 1
|
||
return phrases
|
||
|
||
|
||
def count_in(times: list[float], start: float, end: float) -> int:
|
||
return sum(1 for t in times if t >= start - 1e-6 and t < end - 1e-6)
|
||
|
||
|
||
def derive_phase_budgets(timeline: dict) -> list:
|
||
"""Attach an objective density read to each energy phase.
|
||
|
||
Density is a fact (onsets-per-second + rolls). It is a hint for how much visual
|
||
content a span can hold; it does not set timing or sections. The Music Reader uses
|
||
these phases to decide the actual sections.
|
||
"""
|
||
phases = timeline.get("energy_phases", [])
|
||
onset_times = [e["t"] for e in timeline.get("events", [])]
|
||
rolls = timeline.get("rolls", [])
|
||
hard_stops = timeline.get("hard_stops", [])
|
||
|
||
out = []
|
||
for s in phases:
|
||
span = max(1e-6, float(s.get("end", 0)) - float(s.get("start", 0)))
|
||
onsets = count_in(onset_times, s["start"], s["end"])
|
||
ph_rolls = [
|
||
{"start": r["start"], "end": r["end"], "kind": r["kind"], "drum": r["drum"]}
|
||
for r in rolls
|
||
if r["start"] < s["end"] - 1e-6 and r["end"] > s["start"] + 1e-6
|
||
]
|
||
ph_stops = [
|
||
h["t"]
|
||
for h in hard_stops
|
||
if h["t"] >= s["start"] - 1e-6 and h["t"] < s["end"] + 1e-6
|
||
]
|
||
|
||
if s.get("energy", 0) < 0.2 or onsets < 6:
|
||
density = "sparse"
|
||
elif onsets >= 18 or ph_rolls:
|
||
density = "dense"
|
||
else:
|
||
density = "medium"
|
||
|
||
enriched = dict(s)
|
||
enriched["onsets"] = onsets
|
||
enriched["onsetRate"] = round(onsets / span, 1)
|
||
enriched["rolls"] = ph_rolls
|
||
enriched["hardStops"] = ph_stops
|
||
enriched["density"] = density
|
||
out.append(enriched)
|
||
return out
|
||
|
||
|
||
def finalize_audiomap(timeline: dict, phrase_bars: int = 4) -> dict:
|
||
downbeats = timeline.get("grid", {}).get("downbeats_sec", [])
|
||
duration_sec = timeline.get("audio", {}).get("duration_sec", 0)
|
||
energy_phases = derive_phase_budgets(timeline)
|
||
phrases = derive_phrases(downbeats, phrase_bars, duration_sec)
|
||
return {
|
||
"version": AUDIOMAP_VERSION,
|
||
"phraseBars": phrase_bars,
|
||
**timeline,
|
||
"energy_phases": energy_phases,
|
||
"phrases": phrases,
|
||
}
|
||
|
||
|
||
# ── main ────────────────────────────────────────────────────────────────────
|
||
def analyze(path: str, phrase_bars: int = 4) -> dict:
|
||
y, sr, dur = load_audio(path)
|
||
bg = beat_grid(y, sr)
|
||
bc = band_energy_curves(y, sr)
|
||
phase = downbeat_phase(bg["beat_frames"], bc)
|
||
beats = bg["beats"]
|
||
downbeats = [float(beats[i]) for i in range(len(beats)) if (i - phase) % 4 == 0]
|
||
|
||
# onsets → events
|
||
onset_t = librosa.onset.onset_detect(
|
||
y=y, sr=sr, hop_length=HOP, units="time", backtrack=True
|
||
)
|
||
en_at = bc["total"]
|
||
en_max = float(en_at.max()) + 1e-9
|
||
events = []
|
||
for t in onset_t:
|
||
gclass, bar, bib, step16 = classify_metric(float(t), beats, phase)
|
||
drum, energy, feel = classify_drum(float(t), bc, sr)
|
||
f = frame_at(float(t), sr, bc["n"])
|
||
events.append({
|
||
"t": round(float(t), 3),
|
||
"bar": int(bar), "beat_in_bar": int(bib), "step16": int(step16),
|
||
"grid": gclass, "drum": drum,
|
||
"energy": round(float(en_at[f]) / en_max, 2), "feel": feel,
|
||
"special": None,
|
||
})
|
||
|
||
first_onset = next((float(t) for t in onset_t if t >= 2.0), 0.0)
|
||
es = energy_structure(y, sr, dur, first_onset)
|
||
|
||
# tag specials onto nearby events
|
||
for hs in es["hard_stops"]:
|
||
for e in events:
|
||
if abs(e["t"] - hs["t"]) < 0.6:
|
||
e["special"] = "hard_stop"
|
||
# riser: events inside a 1.5s+ rising-energy run that precedes a SURGE
|
||
surges = [m["t"] for m in es["moments"] if m["kind"] == "SURGE"]
|
||
for st in surges:
|
||
for e in events:
|
||
if st - 2.0 <= e["t"] < st and e["special"] is None and e["drum"] in ("perc", "glitch"):
|
||
e["special"] = "riser"
|
||
|
||
# rolls / fills + whether each leads straight into a surge/drop (cascade cue)
|
||
beat_dur = float(np.median(np.diff(beats))) if len(beats) > 1 else 60.0 / max(bg["bpm"], 1e-6)
|
||
rolls = detect_rolls(events, beat_dur)
|
||
for r in rolls:
|
||
r["leads_to"] = next((m["kind"] for m in es["moments"]
|
||
if 0 <= m["t"] - r["end"] <= 1.2), None)
|
||
# near-silent windows (= VOID energy phases) — convenience for "hold / breathe"
|
||
silences = [{"start": p["start"], "end": p["end"]}
|
||
for p in es["phases"] if p["level"] == "VOID"]
|
||
# per-phase sustained spectral character (how each energy block FEELS)
|
||
annotate_section_feel(bc, sr, es["phases"])
|
||
|
||
n_drum = {}
|
||
for e in events:
|
||
n_drum[e["drum"]] = n_drum.get(e["drum"], 0) + 1
|
||
n_grid = {}
|
||
for e in events:
|
||
n_grid[e["grid"]] = n_grid.get(e["grid"], 0) + 1
|
||
|
||
summary = (f"{bg['bpm']:.0f} BPM · {len(beats)} beats / {len(downbeats)} bars · "
|
||
f"{len(events)} events ({n_drum}) · {len(rolls)} rolls · "
|
||
f"{len(es['phases'])} energy phases · {dur:.1f}s")
|
||
|
||
timeline = {
|
||
"summary": summary,
|
||
"audio": {"path": path, "duration_sec": round(dur, 3), "sr": sr},
|
||
"tempo": {"bpm": round(bg["bpm"], 1), "beats_per_bar": 4,
|
||
"downbeat_phase": phase, "n_beats": len(beats), "n_bars": len(downbeats)},
|
||
"grid": {"beats_sec": [round(float(b), 3) for b in beats],
|
||
"downbeats_sec": [round(b, 3) for b in downbeats]},
|
||
"energy_phases": es["phases"],
|
||
"key_moments": es["moments"],
|
||
"hard_stops": es["hard_stops"],
|
||
"rolls": rolls,
|
||
"silences": silences,
|
||
"stats": {"drum_counts": n_drum, "grid_counts": n_grid},
|
||
"events": events,
|
||
}
|
||
return finalize_audiomap(timeline, phrase_bars)
|
||
|
||
|
||
def print_brief(d: dict) -> None:
|
||
print(f"\n{d['summary']}\n{'='*70}")
|
||
print("ENERGY PHASES (audio-driven blocks; the Music Reader names the sections)")
|
||
for s in d.get("energy_phases", []):
|
||
feel = s.get("feel", {})
|
||
bands = ",".join(feel.get("bands", []))
|
||
print(f" {s['start']:5.1f}-{s['end']:5.1f}s {s.get('level', ''):6s} energy={s.get('energy')} "
|
||
f"{feel.get('character', ''):6s} [{bands}] {s.get('density', '?')}")
|
||
print("PHRASES")
|
||
for p in d.get("phrases", []):
|
||
print(f" #{p['index']} {p['start']:5.2f}-{p['end']:5.2f}s bars={p['bars']}")
|
||
print("KEY MOMENTS")
|
||
for m in d["key_moments"]:
|
||
print(f" {m['t']:3d}s {m['kind']:5s} Δ{m['delta']:+.2f}")
|
||
print(f"HARD STOPS: {[h['t'] for h in d['hard_stops']]}")
|
||
print("ROLLS / FILLS")
|
||
for r in d.get("rolls", []):
|
||
lead = f" → {r['leads_to']}" if r.get("leads_to") else ""
|
||
print(f" {r['start']:6.2f}-{r['end']:5.2f}s {r['hits']:2d} hits @ {r['rate_per_min']:4d}/min "
|
||
f"{r['kind']:10s} ({r['drum']}){lead}")
|
||
print(f"SILENCES: {[(s['start'], s['end']) for s in d.get('silences', [])]}")
|
||
print(f"DRUM COUNTS: {d['stats']['drum_counts']} GRID: {d['stats']['grid_counts']}")
|
||
print(f"\nEVENTS ({len(d['events'])}) [t · bar:beat · grid · drum · energy · special]")
|
||
for e in d["events"]:
|
||
sp = f" <{e['special']}>" if e["special"] else ""
|
||
print(f" {e['t']:6.2f}s b{e['bar']}:{e['beat_in_bar']} {e['grid']:3s} "
|
||
f"{e['drum']:6s} e={e['energy']:.2f} {e['feel']:8s}{sp}")
|
||
|
||
|
||
def main() -> None:
|
||
ap = argparse.ArgumentParser()
|
||
ap.add_argument("audio")
|
||
ap.add_argument("-o", "--out", default=None)
|
||
ap.add_argument("--phrase-bars", type=int, default=4)
|
||
ap.add_argument("--print", action="store_true", dest="do_print")
|
||
a = ap.parse_args()
|
||
d = analyze(a.audio, phrase_bars=a.phrase_bars)
|
||
if a.out:
|
||
# ensure_ascii=False means the payload can carry non-ASCII, so the file
|
||
# encoding cannot be left to the platform default (cp1252 on Windows).
|
||
Path(a.out).write_text(json.dumps(d, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
dens = " ".join(f"{s.get('level', '?')}:{s.get('density', '?')}" for s in d.get("energy_phases", []))
|
||
print(
|
||
f"[analyze-beatgrid] wrote audiomap {a.out} · {len(d.get('energy_phases', []))} phases · density [{dens}]",
|
||
file=sys.stderr,
|
||
)
|
||
if a.do_print or not a.out:
|
||
print_brief(d)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|