Spaces:
Running on Zero
Running on Zero
multimodalart HF Staff
Calibrate ZeroGPU duration from measured per-frame throughput
ca948bd verified Download app.py from Mothersuperior/yue2-hum-to-song: direct link, hf CLI and curl.
- Browser
- Download file 28.6 kB
-
https://huggingface.co/spaces/Mothersuperior/yue2-hum-to-song/resolve/main/app.py
- Command line
-
hf download hf://spaces/Mothersuperior/yue2-hum-to-song/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Mothersuperior/yue2-hum-to-song/resolve/main/app.py
28.6 kB
| """Hum to Song — YuE2-3B + the hum-to-song prosody adapter. | |
| Two stages, exactly as described by the adapter's model card and its reference | |
| scripts (`hum_continue.py` / `infer_hum.py`): | |
| 1. Score continuation: SheetSage2 transcribes the hum to ABC, the score prefix is | |
| left *open* (no `[ABC_END]`), and YuE2's planner keeps writing the song from | |
| the hummed bars, then writes semantic tokens for the whole song. | |
| 2. Prosody adapter: the hum is reduced to a pitch carrier (pYIN f0 -> sine, | |
| amplitude = heavily low-passed |hum|), VAE-encoded, projected by four | |
| `Linear(64 -> 2048)` heads and added to the NAR hidden state at layers | |
| 0 / 7 / 14 / 21 while the flow-matching decoder renders the audio. | |
| """ | |
| import os | |
| os.environ.setdefault("NUMBA_CACHE_DIR", "/tmp/numba_cache") | |
| os.environ.setdefault("NUMBA_DISABLE_CUDA", "1") | |
| os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") | |
| os.makedirs("/tmp/numba_cache", exist_ok=True) | |
| import spaces # noqa: E402 (must precede any CUDA-touching import) | |
| import dataclasses # noqa: E402 | |
| import re # noqa: E402 | |
| import subprocess # noqa: E402 | |
| import tempfile # noqa: E402 | |
| import time # noqa: E402 | |
| from pathlib import Path # noqa: E402 | |
| import gradio as gr # noqa: E402 | |
| import numpy as np # noqa: E402 | |
| import soundfile as sf # noqa: E402 | |
| import torch # noqa: E402 | |
| import torch.nn as nn # noqa: E402 | |
| import torch.nn.functional as F # noqa: E402 | |
| from huggingface_hub import hf_hub_download # noqa: E402 | |
| from safetensors.torch import load_file # noqa: E402 | |
| from transformers import AutoModel # noqa: E402 | |
| from yue2 import YuE2Pipeline # noqa: E402 | |
| from yue2 import nar as yue2_nar # noqa: E402 | |
| from yue2.modeling_vae import YuE2VAE # noqa: E402 | |
| from yue2.modeling_yue2 import YuE2ForCausalLM # noqa: E402 | |
| from yue2.pipeline import SymbolicPlan # noqa: E402 | |
| from yue2.protocol import EOD, SongRequest, resolve_sampling, token_prefixes # noqa: E402 | |
| MODEL_ID = "m-a-p/YuE2-3B" | |
| VAE_ID = "m-a-p/YuE2-Vae" | |
| TRANSCRIBER_ID = "m-a-p/SheetSage2" | |
| ADAPTER_REPO = "Mothersuperior/YuE2-hum-to-song" | |
| ADAPTER_FILE = "hum_adapter_v1_combined.safetensors" | |
| INJECT_LAYERS = [0, 7, 14, 21] # documented in the adapter's safetensors metadata | |
| SR = 48_000 | |
| HOP = 512 | |
| FRAME_RATE = 25 # VAE latent frames per second | |
| MAX_HUM_SECONDS = 30 | |
| # -------------------------------------------------------------------------------------- | |
| # Load everything once, at module scope, and move it to CUDA eagerly (ZeroGPU packs the | |
| # weights to disk here and streams them into VRAM on the first @spaces.GPU call). | |
| # -------------------------------------------------------------------------------------- | |
| # YuE2's process-wide memory-fraction cap forces a real CUDA init, which cannot happen | |
| # in ZeroGPU's emulated startup environment; skip only that call while constructing. | |
| _set_fraction = torch.cuda.set_per_process_memory_fraction | |
| try: | |
| torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None | |
| pipe = YuE2Pipeline.from_pretrained( | |
| MODEL_ID, vae=VAE_ID, device="cuda", backend="torch", | |
| verify_hashes=False, progress=False, | |
| ) | |
| finally: | |
| torch.cuda.set_per_process_memory_fraction = _set_fraction | |
| tokenizer = pipe.tokenizer | |
| model = YuE2ForCausalLM.from_pretrained( | |
| pipe.model_dir, local_files_only=True, torch_dtype=torch.bfloat16, low_cpu_mem_usage=True | |
| ).eval() | |
| _adapter = load_file(hf_hub_download(ADAPTER_REPO, ADAPTER_FILE)) | |
| _HIDDEN = model.config.hidden_size | |
| hum_proj = nn.ModuleList([nn.Linear(64, _HIDDEN) for _ in INJECT_LAYERS]) | |
| with torch.no_grad(): | |
| # rank-96 LoRA folded into the NAR (decoder) branch only; the AR branch that writes | |
| # the score and the semantic tokens is left bit-identical to stock YuE2-3B. | |
| for index, layer in enumerate(model.model.layers): | |
| for module, group, names in ( | |
| (layer.nar_self_attn, "nar_self_attn", ("q_proj", "k_proj", "v_proj", "o_proj")), | |
| (layer.nar_mlp, "nar_mlp", ("gate_proj", "up_proj", "down_proj")), | |
| ): | |
| for name in names: | |
| key = f"layers.{index}.{group}.{name}" | |
| delta = _adapter[f"{key}.lora_B"].float() @ _adapter[f"{key}.lora_A"].float() | |
| linear = getattr(module, name) | |
| linear.weight.add_(delta.to(linear.weight.dtype)) | |
| # vae2llm / llm2vae ship as full replacements | |
| for target, prefix in ((model.vae2llm, "vae2llm"), (model.llm2vae, "llm2vae")): | |
| target.weight.copy_(_adapter[f"{prefix}.weight"].to(target.weight.dtype)) | |
| target.bias.copy_(_adapter[f"{prefix}.bias"].to(target.bias.dtype)) | |
| for slot in range(len(INJECT_LAYERS)): | |
| hum_proj[slot].weight.copy_(_adapter[f"hum_proj.{slot}.weight"]) | |
| hum_proj[slot].bias.copy_(_adapter[f"hum_proj.{slot}.bias"]) | |
| del _adapter | |
| hum_proj = hum_proj.to(torch.bfloat16).eval() | |
| pipe._model = model | |
| model.to("cuda") | |
| hum_proj.to("cuda") | |
| # full VAE (the carrier has to be *encoded*, the song *decoded*) | |
| vae = YuE2VAE.from_pretrained(pipe.vae_dir, decoder_only=False, device="cpu", local_files_only=True) | |
| vae.to("cuda") | |
| transcriber = AutoModel.from_pretrained(TRANSCRIBER_ID, trust_remote_code=True).eval() | |
| transcriber.to("cuda") | |
| # -------------------------------------------------------------------------------------- | |
| # Hum -> pitch carrier (verbatim from the adapter's hum_prep.py) | |
| # -------------------------------------------------------------------------------------- | |
| def load_audio(path: str, sr_out: int = SR, mono: bool = True) -> np.ndarray: | |
| """Decode any audio file to float32 PCM with ffmpeg, matching the training prep.""" | |
| result = subprocess.run( | |
| ["ffmpeg", "-v", "error", "-i", str(path), "-f", "f32le", | |
| "-ac", "1" if mono else "2", "-ar", str(sr_out), "-"], | |
| capture_output=True, | |
| ) | |
| audio = np.frombuffer(result.stdout, dtype=np.float32) | |
| if audio.size == 0: | |
| raise gr.Error("Could not decode that audio file. Please upload a wav/mp3/flac hum.") | |
| return audio if mono else audio.reshape(-1, 2) | |
| def prosody_sine(y: np.ndarray, sr: int = SR): | |
| """Reduce a hum to melody + timing only: pYIN f0 -> sine, low-passed |y| envelope.""" | |
| import librosa | |
| from scipy.signal import butter, sosfiltfilt | |
| f0, voiced, _ = librosa.pyin(y, fmin=65, fmax=1000, sr=sr, hop_length=HOP, frame_length=4 * HOP) | |
| f0 = f0.copy() | |
| index = np.where(~np.isnan(f0))[0] | |
| if len(index) == 0: | |
| return None, 0.0 | |
| f0[: index[0]] = f0[index[0]] | |
| for i in range(1, len(f0)): | |
| if np.isnan(f0[i]): | |
| f0[i] = f0[i - 1] | |
| t = np.arange(len(y)) / sr | |
| frequency = np.interp(t, np.arange(len(f0)) * HOP / sr, f0) | |
| envelope = np.abs(y) | |
| envelope = sosfiltfilt(butter(4, 30, btype="low", fs=sr, output="sos"), envelope) | |
| envelope = sosfiltfilt(butter(2, 80, btype="low", fs=sr, output="sos"), envelope) | |
| envelope = np.clip(envelope, 0, None) | |
| carrier = envelope * np.sin(2 * np.pi * np.cumsum(frequency) / sr) | |
| carrier = (carrier / (np.abs(carrier).max() + 1e-9) * 0.9).astype(np.float32) | |
| return carrier, float(np.nanmean(voiced)) | |
| # Compile pYIN's numba kernels at startup rather than inside the first GPU call. | |
| try: | |
| prosody_sine(np.sin(np.linspace(0, 2 * np.pi * 220 * 2, SR * 2)).astype(np.float32)) | |
| except Exception as _warmup_error: # noqa: BLE001 | |
| print(f"pYIN warmup skipped: {_warmup_error}", flush=True) | |
| def carrier_latents(carrier: np.ndarray) -> torch.Tensor: | |
| """VAE-encode the stereo-duplicated carrier to [frames, 64] latents at 25 Hz.""" | |
| stereo = np.stack([carrier, carrier], 1) | |
| chunk = SR * 30 | |
| pieces = [] | |
| for start in range(0, len(stereo), chunk): | |
| segment = stereo[start:start + chunk] | |
| if len(segment) < 1920: | |
| break | |
| tensor = torch.tensor(np.ascontiguousarray(segment.T[None])) | |
| pieces.append(vae.encode(tensor)[0].T.float()) | |
| if not pieces: | |
| raise gr.Error("The hum is too short — please record at least a second of melody.") | |
| return torch.cat(pieces, 0) | |
| # -------------------------------------------------------------------------------------- | |
| # Score helpers | |
| # -------------------------------------------------------------------------------------- | |
| _VOICE = re.compile(r"^V:\s*(\w+)") | |
| _RESTS = re.compile(r"(Z\d*\|)+") | |
| _TRAILING = re.compile(r"(V: (Vocal|Ins))|(Z\d*\|)|") | |
| def promote_hum_to_vocal(abc: str): | |
| """Put the hummed melody on the Vocal staff. | |
| SheetSage2 has no words to latch onto in a hum, so it usually files the melody | |
| under the `Ins` voice and leaves `Vocal` as whole-bar rests. A hum *is* the vocal | |
| line, so when Vocal is rests-only the two staves are swapped before the planner | |
| continues the score. | |
| """ | |
| lines = abc.split("\n") | |
| key_index = next((i for i, line in enumerate(lines) if line.startswith("K:")), None) | |
| if key_index is None: | |
| return abc, False | |
| head, body = lines[: key_index + 1], lines[key_index + 1:] | |
| blocks, current = [], None | |
| for line in body: | |
| stripped = line.strip() | |
| match = _VOICE.match(stripped) | |
| if match: | |
| current = {"voice": match.group(1), "header": line, "lines": []} | |
| blocks.append(current) | |
| elif not stripped or stripped.startswith("%"): | |
| blocks.append({"voice": None, "header": line, "lines": []}) | |
| current = None | |
| elif current is not None: | |
| current["lines"].append(line) | |
| else: | |
| blocks.append({"voice": None, "header": line, "lines": []}) | |
| vocal = [b for b in blocks if b["voice"] == "Vocal"] | |
| ins = [b for b in blocks if b["voice"] == "Ins"] | |
| def rests_only(group): | |
| text = "".join(line.strip() for block in group for line in block["lines"]) | |
| return bool(text) and _RESTS.fullmatch(text) is not None | |
| if not ins or not vocal or not rests_only(vocal) or rests_only(ins): | |
| return abc, False | |
| for a, b in zip(vocal, ins): | |
| a["lines"], b["lines"] = b["lines"], a["lines"] | |
| out = list(head) | |
| for block in blocks: | |
| out.append(block["header"]) | |
| out.extend(block["lines"]) | |
| return "\n".join(out) + "\n", True | |
| def open_score(abc: str) -> str: | |
| """Drop trailing rest-only bars / dangling voice markers so the score stays open.""" | |
| lines = abc.split("\n") | |
| while lines and _TRAILING.fullmatch(lines[-1].strip()): | |
| lines.pop() | |
| return "\n".join(lines) + "\n" | |
| # -------------------------------------------------------------------------------------- | |
| # Hum-conditioned flow-matching decoder | |
| # -------------------------------------------------------------------------------------- | |
| class HumCachedNAR(yue2_nar.CachedNAR): | |
| """CachedNAR with the carrier projections added at layers 0 / 7 / 14 / 21.""" | |
| def __init__(self, model, chunk, carrier, guidance=1.0, **kwargs): | |
| super().__init__(model, chunk, **kwargs) | |
| padded = F.pad(carrier, (0, 0, 1, 1))[None].to(self.dtype) | |
| self.hum = {layer: hum_proj[slot](padded) for slot, layer in enumerate(INJECT_LAYERS)} | |
| self.guidance = float(guidance) | |
| self.blank = None | |
| if self.guidance != 1.0: | |
| zeros = torch.zeros_like(padded) | |
| self.blank = {layer: hum_proj[slot](zeros) for slot, layer in enumerate(INJECT_LAYERS)} | |
| def _branch(self, state, raw_t, injection): | |
| model = self.model | |
| shifted = model._shift_t_value(raw_t, self.device, self.dtype) | |
| x = model.vae2llm(F.pad(state, (0, 0, 1, 1))[None].to(self.dtype)) | |
| x = x + injection[0] | |
| x = x + model.time_embedder(shifted.expand(self.nar_length))[None] | |
| x = x + self.pos_emb | |
| for index, (layer, (ar_k, ar_v)) in enumerate(zip(model.model.layers, self.cache)): | |
| if index and index in injection: | |
| x = x + injection[index] | |
| q, k, v = layer.nar_self_attn.project_qkv(layer.nar_input_layernorm(x), self.cos, self.sin) | |
| k, v = torch.cat((ar_k, k[0])), torch.cat((ar_v, v[0])) | |
| h = self._attention(q[0], k, v) | |
| x = x + layer.nar_self_attn.o_proj(h.flatten(1)[None]) | |
| x = x + layer.nar_mlp(layer.nar_pre_mlp_layernorm(x)) | |
| return model.llm2vae(model.model.norm(x))[0, 1:-1].float() | |
| def velocity(self, state, raw_t): | |
| conditioned = self._branch(state, raw_t, self.hum) | |
| if self.blank is None: | |
| return conditioned | |
| unconditioned = self._branch(state, raw_t, self.blank) | |
| return unconditioned + self.guidance * (conditioned - unconditioned) | |
| def solve(self, steps=32, cancelled=None, on_progress=None): | |
| """Reference 32-step midpoint solver, integrated in FP32 as in infer_hum.py.""" | |
| state = self.chunk.noise.to(device=self.device) | |
| dt = 1.0 / int(steps) | |
| for step in range(int(steps)): | |
| t = 1.0 - step * dt | |
| raw = torch.logit(torch.tensor(t, dtype=torch.float64)).clamp(-20, 20).item() | |
| first = self.velocity(state, raw) | |
| mid = state - first * (dt / 2) | |
| raw_mid = torch.logit(torch.tensor(t - dt / 2, dtype=torch.float64)).clamp(-20, 20).item() | |
| state = state - self.velocity(mid, raw_mid) * dt | |
| result = state.float().cpu() | |
| if not torch.isfinite(result).all(): | |
| raise gr.Error("The decoder produced non-finite latents; try another seed.") | |
| return result | |
| def hum_synthesize(prefix, codec, condition, seed, steps, guidance): | |
| chunks = yue2_nar.song_chunks(prefix, codec, seed, pipe.generation_config.context) | |
| pieces, offset = [], 0 | |
| for chunk in chunks: | |
| frames = len(chunk.noise) | |
| engine = HumCachedNAR(model, chunk, condition[offset:offset + frames], guidance=guidance) | |
| try: | |
| pieces.append(engine.solve(steps)) | |
| finally: | |
| engine.close() | |
| offset += frames | |
| del engine | |
| return torch.cat(pieces, 0) | |
| MELODY_MODES = { | |
| "Continue from hum": "continue", | |
| "Hum only": "hum_only", | |
| "Ignore hum melody": "ignore", | |
| } | |
| DEFAULT_STYLE = "City Pop, upbeat, danceable, groovy bass, electric guitar, synth, energetic, joyful, neon city night" | |
| DEFAULT_LYRICS = """[Intro] | |
| [Verse] | |
| Neon on the window, humming to the rain | |
| Every empty station calls my name again | |
| Take the midnight highway, leave the quiet town | |
| Turn the radio over, turn the feeling down | |
| [Chorus] | |
| Hold the night a little longer | |
| Every heartbeat getting stronger | |
| Sing it back to me | |
| Sing it back to me | |
| [Outro] | |
| Sing it back to me | |
| """ | |
| def estimate_duration(hum=None, style=None, lyrics=None, melody_mode="Continue from hum", | |
| max_seconds=140, hum_influence=1.0, ode_steps=32, seed=831001, | |
| *args, **kwargs): | |
| """Per-frame cost model fitted to measured runs on this Space's hardware. | |
| Measured (frames -> seconds): semantic 0.0095/frame, NAR decode | |
| 0.000088/frame/ODE-step/branch, VAE decode 0.00042/frame; pYIN carrier | |
| 0.25 s per second of hum. The song length is decided by the planner, so | |
| this budgets for the worst case where it fills ``max_seconds``. | |
| """ | |
| hum_seconds = 0.0 | |
| try: | |
| path = hum.get("path") if isinstance(hum, dict) else hum | |
| if path: | |
| info = sf.info(str(path)) | |
| hum_seconds = min(MAX_HUM_SECONDS, info.frames / max(1, info.samplerate)) | |
| except Exception: | |
| hum_seconds = 0.0 | |
| hum_seconds = hum_seconds or MAX_HUM_SECONDS | |
| steps = max(4, int(ode_steps or 32)) | |
| branches = 2.0 if float(hum_influence or 1.0) != 1.0 else 1.0 | |
| frames = 25.0 * float(max_seconds or 140) | |
| fixed = 6.0 + 0.25 * hum_seconds + 0.15 * hum_seconds # write/overhead + pYIN + transcribe | |
| variable = frames * (0.0095 + 0.000088 * steps * branches + 0.00042) | |
| return int(min(300, round(1.2 * (fixed + variable)))) | |
| def hum_to_song(hum, style: str = DEFAULT_STYLE, lyrics: str = DEFAULT_LYRICS, | |
| melody_mode: str = "Continue from hum", max_seconds: int = 140, | |
| hum_influence: float = 1.0, ode_steps: int = 32, seed: int = 831001): | |
| """Turn a hummed melody plus a style line and lyrics into a finished song. | |
| Args: | |
| hum: path to a 10-30 s recording of a hummed melody (voice, not a synth tone). | |
| style: comma-separated style / instrumentation tags for the arrangement. | |
| lyrics: section-tagged lyrics, e.g. "[Verse] ... [Chorus] ...". | |
| melody_mode: "Continue from hum" lets the planner extend your melody into a | |
| full song, "Hum only" keeps exactly your melody, "Ignore hum melody" lets | |
| the planner write its own tune and uses the hum for phrasing only. | |
| max_seconds: hard cap on the rendered song length. | |
| hum_influence: classifier-free guidance on the hum channel (1.0 = as trained). | |
| ode_steps: flow-matching midpoint steps for the decoder. | |
| seed: sampling seed. | |
| Returns: | |
| The song as MP3, the same song as 24-bit FLAC, the pitch carrier the decoder | |
| was conditioned on, the ABC score, and a run summary. | |
| """ | |
| if not hum: | |
| raise gr.Error("Please record or upload a hummed melody first.") | |
| style = (style or "").strip() | |
| lyrics = (lyrics or "").strip() | |
| if not style or not lyrics: | |
| raise gr.Error("Please provide both a style line and lyrics.") | |
| pipe.device = next(model.parameters()).device | |
| mode = MELODY_MODES.get(melody_mode, "continue") | |
| seed = int(seed) % (2 ** 31) | |
| steps = max(4, int(ode_steps)) | |
| guidance = float(hum_influence) | |
| timings = {} | |
| started = time.perf_counter() | |
| # ---- hum -> carrier ------------------------------------------------------------- | |
| mark = time.perf_counter() | |
| audio = load_audio(hum, SR, True)[: MAX_HUM_SECONDS * SR] | |
| if len(audio) < SR: | |
| raise gr.Error("That clip is under a second — please hum for 10 to 30 seconds.") | |
| carrier, voiced = prosody_sine(audio, SR) | |
| if carrier is None or voiced < 0.02: | |
| raise gr.Error("No pitch was found in that clip. Hum a melody with your voice, " | |
| "close to the microphone.") | |
| timings["carrier"] = time.perf_counter() - mark | |
| # ---- hum -> ABC ------------------------------------------------------------------ | |
| mark = time.perf_counter() | |
| swapped = False | |
| hum_abc = None | |
| if mode != "ignore": | |
| try: | |
| result = transcriber.transcribe(audio[None], sampling_rate=SR, melody_only=True) | |
| hum_abc = result.get("abc") | |
| except Exception as error: # noqa: BLE001 | |
| raise gr.Error(f"Melody transcription failed: {error}") from error | |
| if not hum_abc or not hum_abc.strip(): | |
| raise gr.Error("The melody transcriber could not read that clip. Try a clearer, " | |
| "steadier hum (it needs a voice, not a synthesised tone).") | |
| hum_abc, swapped = promote_hum_to_vocal(hum_abc) | |
| timings["transcribe"] = time.perf_counter() - mark | |
| # ---- stage 1: plan the score ----------------------------------------------------- | |
| mark = time.perf_counter() | |
| request = SongRequest(style=style, lyrics=lyrics, cot="melody", seed=seed, id="hum") | |
| if mode == "continue": | |
| partial = tokenizer.encode(open_score(hum_abc)) | |
| prefix = token_prefixes(request, tokenizer) + partial | |
| sampling = resolve_sampling(None, pipe.generation_config.abc) | |
| ids, timing, truncated = pipe._generate(prefix, sampling, seed, "abc") | |
| abc_ids = partial + [int(t) for t in ids if int(t) < EOD] | |
| plan = SymbolicPlan(request, tokenizer.decode(abc_ids), abc_ids, | |
| token_prefixes(request, tokenizer, abc_ids), timing, truncated) | |
| elif mode == "hum_only": | |
| request = dataclasses.replace(request, abc=hum_abc) | |
| plan = pipe.plan(request=request) | |
| else: | |
| plan = pipe.plan(request=request) | |
| timings["score"] = time.perf_counter() - mark | |
| # ---- stage 1b: semantic tokens --------------------------------------------------- | |
| mark = time.perf_counter() | |
| cap = max(400, int(float(max_seconds) * FRAME_RATE)) | |
| headroom = pipe.generation_config.context - len(plan.prefix) - 8 | |
| if headroom < 400: | |
| raise gr.Error("Those lyrics plus the planned score fill the model's context. " | |
| "Please shorten the lyrics.") | |
| cap = min(cap, headroom) | |
| semantic = pipe.generate_semantic(plan, sampling=resolve_sampling({"max_tokens": cap}, | |
| pipe.generation_config.semantic)) | |
| codec = list(semantic.tokens) | |
| if not codec: | |
| raise gr.Error("The planner produced no audio tokens; try another seed.") | |
| timings["semantic"] = time.perf_counter() - mark | |
| # ---- stage 2: hum-conditioned decode -------------------------------------------- | |
| mark = time.perf_counter() | |
| latents = carrier_latents(carrier) | |
| frames = len(codec) | |
| condition = torch.zeros(frames, 64, dtype=torch.float32, device=latents.device) | |
| used = min(len(latents), frames) | |
| condition[:used] = latents[:used] | |
| timings["encode"] = time.perf_counter() - mark | |
| mark = time.perf_counter() | |
| z = hum_synthesize(plan.prefix, codec, condition, seed, steps, guidance) | |
| timings["decoder"] = time.perf_counter() - mark | |
| mark = time.perf_counter() | |
| with torch.inference_mode(): | |
| wave = vae.decode_tiled(z.T[None].contiguous(), core_frames=750, halo_frames=16, | |
| output_device="cpu") | |
| song = np.nan_to_num(wave[0].float().clamp(-1, 1).T.numpy()).astype(np.float32) | |
| timings["vae"] = time.perf_counter() - mark | |
| # ---- deliver --------------------------------------------------------------------- | |
| out = Path(tempfile.mkdtemp(prefix="humsong_")) | |
| flac_path = out / "song.flac" | |
| mp3_path = out / "song.mp3" | |
| carrier_path = out / "pitch_carrier.mp3" | |
| sf.write(flac_path, song, SR, subtype="PCM_24") | |
| sf.write(out / "carrier.wav", carrier, SR) | |
| for source, target in ((flac_path, mp3_path), (out / "carrier.wav", carrier_path)): | |
| subprocess.run(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source), | |
| "-codec:a", "libmp3lame", "-b:a", "192k", str(target)], | |
| check=True, capture_output=True) | |
| total = time.perf_counter() - started | |
| seconds = len(song) / SR | |
| notes = [ | |
| f"**{seconds:.1f} s** of audio in **{total:.0f} s** on ZeroGPU.", | |
| f"mode `{mode}` · {steps} ODE steps · hum influence {guidance:g} · seed {seed}", | |
| f"hum: {len(audio) / SR:.1f} s, {voiced * 100:.0f}% voiced → " | |
| f"{len(latents)} carrier frames over {frames} song frames", | |
| "stage timings (s): " + " · ".join(f"{k} {v:.1f}" for k, v in timings.items()), | |
| ] | |
| if swapped: | |
| notes.append("_The transcriber filed the hum under the instrumental staff, so it was " | |
| "moved onto the vocal staff before planning._") | |
| if semantic.truncated: | |
| notes.append("_Hit the length cap — raise **max song length** for a full ending._") | |
| return str(mp3_path), str(flac_path), str(carrier_path), plan.abc or "", "\n\n".join(notes) | |
| # -------------------------------------------------------------------------------------- | |
| # UI | |
| # -------------------------------------------------------------------------------------- | |
| CSS = """ | |
| #col-container { max-width: 1100px; margin: 0 auto; } | |
| .dark .gradio-container { color: var(--body-text-color); } | |
| """ | |
| WALTZ_STYLE = ("Indie folk waltz, 3/4, warm female lead vocal, fingerpicked acoustic guitar, " | |
| "upright bass, brushed drums, intimate, nostalgic") | |
| WALTZ_LYRICS = """[Intro] | |
| [Verse] | |
| Count the streetlights one two three | |
| Slowly turning back to me | |
| Every window keeps a song | |
| Something I have hummed along | |
| [Chorus] | |
| Carry me around again | |
| Slow and easy like it was then | |
| Carry me around again | |
| [Outro] | |
| Slowly turning back to me | |
| """ | |
| JAZZ_STYLE = "Jazz-funk, warm lead vocal, Rhodes piano, electric bass, tight drums, brass stabs" | |
| JAZZ_LYRICS = """[Intro] | |
| [Verse] | |
| Morning on the avenue, coffee going cold | |
| Somebody is playing something seventy years old | |
| Out here on the corner where the traffic learns to swing | |
| I have got a little melody and nothing else to bring | |
| [Chorus] | |
| So play it like you mean it | |
| Play it till the daylight ends | |
| Play it like you mean it | |
| [Outro] | |
| Play it till the daylight ends | |
| """ | |
| EXAMPLES = [ | |
| ["examples/hum_02.wav", DEFAULT_STYLE, DEFAULT_LYRICS], | |
| ["examples/hum_01.wav", WALTZ_STYLE, WALTZ_LYRICS], | |
| ["examples/hum_03.wav", JAZZ_STYLE, JAZZ_LYRICS], | |
| ] | |
| with gr.Blocks(title="Hum to Song") as demo: | |
| with gr.Column(elem_id="col-container"): | |
| gr.Markdown( | |
| """ | |
| # 🎤 Hum to Song | |
| Hum a melody for 10–30 seconds, add a style line and lyrics, and get a produced | |
| song that keeps your tune, builds a structure around it and carries on long after | |
| the hum stops. | |
| [`Mothersuperior/YuE2-hum-to-song`](https://huggingface.co/Mothersuperior/YuE2-hum-to-song) | |
| on top of [`m-a-p/YuE2-3B`](https://huggingface.co/m-a-p/YuE2-3B): the hum is | |
| transcribed with [SheetSage2](https://huggingface.co/m-a-p/SheetSage2), the planner | |
| *continues* that open score into a whole song, and a rank-96 adapter feeds your | |
| actual pitch contour into the flow-matching decoder. Expect one to two minutes. | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(): | |
| hum = gr.Audio(label="Your hum (10–30 s)", type="filepath", | |
| sources=["microphone", "upload"]) | |
| style = gr.Textbox(label="Style", value=DEFAULT_STYLE, lines=2, | |
| info="Genre, mood and instrumentation tags.") | |
| lyrics = gr.Textbox(label="Lyrics", value=DEFAULT_LYRICS, lines=12, | |
| info="Use [Intro] / [Verse] / [Chorus] / [Outro] section tags.") | |
| melody_mode = gr.Radio( | |
| label="Melody mode", choices=list(MELODY_MODES), value="Continue from hum", | |
| info="Continue = your melody opens the song and the planner writes the rest.", | |
| ) | |
| run = gr.Button("Make the song", variant="primary") | |
| with gr.Accordion("Advanced", open=False): | |
| max_seconds = gr.Slider(60, 200, value=140, step=10, | |
| label="Max song length (s)") | |
| hum_influence = gr.Slider(1.0, 3.0, value=1.0, step=0.1, | |
| label="Hum influence (CFG on the hum channel)", | |
| info="1.0 = as trained. Above 1.0 doubles decode time.") | |
| ode_steps = gr.Slider(8, 48, value=32, step=4, label="Decoder ODE steps") | |
| seed = gr.Slider(0, 2 ** 31 - 1, value=831001, step=1, label="Seed") | |
| with gr.Column(): | |
| song_out = gr.Audio(label="Song", type="filepath") | |
| info_out = gr.Markdown() | |
| flac_out = gr.File(label="24-bit FLAC") | |
| carrier_out = gr.Audio(label="Pitch carrier — all the decoder sees of your hum", | |
| type="filepath") | |
| score_out = gr.Code(label="Score (ABC)", lines=18) | |
| gr.Examples( | |
| examples=EXAMPLES, | |
| inputs=[hum, style, lyrics], | |
| outputs=[song_out, flac_out, carrier_out, score_out, info_out], | |
| fn=hum_to_song, | |
| cache_examples=True, | |
| cache_mode="lazy", | |
| label="Example hums (CHAD, CC BY-NC 4.0)", | |
| ) | |
| gr.Markdown( | |
| "Weights derive from YuE2-3B and are **CC BY-NC 4.0 — non-commercial use only**. " | |
| "Example hums come from the " | |
| "[CHAD hummings subset](https://huggingface.co/datasets/amanteur/CHAD_hummings) " | |
| "(Amatov et al., ISMIR 2023), CC BY-NC 4.0." | |
| ) | |
| run.click( | |
| fn=hum_to_song, | |
| inputs=[hum, style, lyrics, melody_mode, max_seconds, hum_influence, ode_steps, seed], | |
| outputs=[song_out, flac_out, carrier_out, score_out, info_out], | |
| api_name="hum_to_song", | |
| ) | |
| demo.queue(max_size=12).launch( | |
| theme=gr.themes.Citrus(), css=CSS, mcp_server=True, show_error=True | |
| ) | |