"""Hum to Song — YuE2-3B + the hum-to-song prosody adapter. Two stages, exactly as described by the adapter's model card and its reference scripts (`hum_continue.py` / `infer_hum.py`): 1. Score continuation: SheetSage2 transcribes the hum to ABC, the score prefix is left *open* (no `[ABC_END]`), and YuE2's planner keeps writing the song from the hummed bars, then writes semantic tokens for the whole song. 2. Prosody adapter: the hum is reduced to a pitch carrier (pYIN f0 -> sine, amplitude = heavily low-passed |hum|), VAE-encoded, projected by four `Linear(64 -> 2048)` heads and added to the NAR hidden state at layers 0 / 7 / 14 / 21 while the flow-matching decoder renders the audio. """ import os os.environ.setdefault("NUMBA_CACHE_DIR", "/tmp/numba_cache") os.environ.setdefault("NUMBA_DISABLE_CUDA", "1") os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") os.makedirs("/tmp/numba_cache", exist_ok=True) import spaces # noqa: E402 (must precede any CUDA-touching import) import dataclasses # noqa: E402 import re # noqa: E402 import subprocess # noqa: E402 import tempfile # noqa: E402 import time # noqa: E402 from pathlib import Path # noqa: E402 import gradio as gr # noqa: E402 import numpy as np # noqa: E402 import soundfile as sf # noqa: E402 import torch # noqa: E402 import torch.nn as nn # noqa: E402 import torch.nn.functional as F # noqa: E402 from huggingface_hub import hf_hub_download # noqa: E402 from safetensors.torch import load_file # noqa: E402 from transformers import AutoModel # noqa: E402 from yue2 import YuE2Pipeline # noqa: E402 from yue2 import nar as yue2_nar # noqa: E402 from yue2.modeling_vae import YuE2VAE # noqa: E402 from yue2.modeling_yue2 import YuE2ForCausalLM # noqa: E402 from yue2.pipeline import SymbolicPlan # noqa: E402 from yue2.protocol import EOD, SongRequest, resolve_sampling, token_prefixes # noqa: E402 MODEL_ID = "m-a-p/YuE2-3B" VAE_ID = "m-a-p/YuE2-Vae" TRANSCRIBER_ID = "m-a-p/SheetSage2" ADAPTER_REPO = "Mothersuperior/YuE2-hum-to-song" ADAPTER_FILE = "hum_adapter_v1_combined.safetensors" INJECT_LAYERS = [0, 7, 14, 21] # documented in the adapter's safetensors metadata SR = 48_000 HOP = 512 FRAME_RATE = 25 # VAE latent frames per second MAX_HUM_SECONDS = 30 # -------------------------------------------------------------------------------------- # Load everything once, at module scope, and move it to CUDA eagerly (ZeroGPU packs the # weights to disk here and streams them into VRAM on the first @spaces.GPU call). # -------------------------------------------------------------------------------------- # YuE2's process-wide memory-fraction cap forces a real CUDA init, which cannot happen # in ZeroGPU's emulated startup environment; skip only that call while constructing. _set_fraction = torch.cuda.set_per_process_memory_fraction try: torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None pipe = YuE2Pipeline.from_pretrained( MODEL_ID, vae=VAE_ID, device="cuda", backend="torch", verify_hashes=False, progress=False, ) finally: torch.cuda.set_per_process_memory_fraction = _set_fraction tokenizer = pipe.tokenizer model = YuE2ForCausalLM.from_pretrained( pipe.model_dir, local_files_only=True, torch_dtype=torch.bfloat16, low_cpu_mem_usage=True ).eval() _adapter = load_file(hf_hub_download(ADAPTER_REPO, ADAPTER_FILE)) _HIDDEN = model.config.hidden_size hum_proj = nn.ModuleList([nn.Linear(64, _HIDDEN) for _ in INJECT_LAYERS]) with torch.no_grad(): # rank-96 LoRA folded into the NAR (decoder) branch only; the AR branch that writes # the score and the semantic tokens is left bit-identical to stock YuE2-3B. for index, layer in enumerate(model.model.layers): for module, group, names in ( (layer.nar_self_attn, "nar_self_attn", ("q_proj", "k_proj", "v_proj", "o_proj")), (layer.nar_mlp, "nar_mlp", ("gate_proj", "up_proj", "down_proj")), ): for name in names: key = f"layers.{index}.{group}.{name}" delta = _adapter[f"{key}.lora_B"].float() @ _adapter[f"{key}.lora_A"].float() linear = getattr(module, name) linear.weight.add_(delta.to(linear.weight.dtype)) # vae2llm / llm2vae ship as full replacements for target, prefix in ((model.vae2llm, "vae2llm"), (model.llm2vae, "llm2vae")): target.weight.copy_(_adapter[f"{prefix}.weight"].to(target.weight.dtype)) target.bias.copy_(_adapter[f"{prefix}.bias"].to(target.bias.dtype)) for slot in range(len(INJECT_LAYERS)): hum_proj[slot].weight.copy_(_adapter[f"hum_proj.{slot}.weight"]) hum_proj[slot].bias.copy_(_adapter[f"hum_proj.{slot}.bias"]) del _adapter hum_proj = hum_proj.to(torch.bfloat16).eval() pipe._model = model model.to("cuda") hum_proj.to("cuda") # full VAE (the carrier has to be *encoded*, the song *decoded*) vae = YuE2VAE.from_pretrained(pipe.vae_dir, decoder_only=False, device="cpu", local_files_only=True) vae.to("cuda") transcriber = AutoModel.from_pretrained(TRANSCRIBER_ID, trust_remote_code=True).eval() transcriber.to("cuda") # -------------------------------------------------------------------------------------- # Hum -> pitch carrier (verbatim from the adapter's hum_prep.py) # -------------------------------------------------------------------------------------- def load_audio(path: str, sr_out: int = SR, mono: bool = True) -> np.ndarray: """Decode any audio file to float32 PCM with ffmpeg, matching the training prep.""" result = subprocess.run( ["ffmpeg", "-v", "error", "-i", str(path), "-f", "f32le", "-ac", "1" if mono else "2", "-ar", str(sr_out), "-"], capture_output=True, ) audio = np.frombuffer(result.stdout, dtype=np.float32) if audio.size == 0: raise gr.Error("Could not decode that audio file. Please upload a wav/mp3/flac hum.") return audio if mono else audio.reshape(-1, 2) def prosody_sine(y: np.ndarray, sr: int = SR): """Reduce a hum to melody + timing only: pYIN f0 -> sine, low-passed |y| envelope.""" import librosa from scipy.signal import butter, sosfiltfilt f0, voiced, _ = librosa.pyin(y, fmin=65, fmax=1000, sr=sr, hop_length=HOP, frame_length=4 * HOP) f0 = f0.copy() index = np.where(~np.isnan(f0))[0] if len(index) == 0: return None, 0.0 f0[: index[0]] = f0[index[0]] for i in range(1, len(f0)): if np.isnan(f0[i]): f0[i] = f0[i - 1] t = np.arange(len(y)) / sr frequency = np.interp(t, np.arange(len(f0)) * HOP / sr, f0) envelope = np.abs(y) envelope = sosfiltfilt(butter(4, 30, btype="low", fs=sr, output="sos"), envelope) envelope = sosfiltfilt(butter(2, 80, btype="low", fs=sr, output="sos"), envelope) envelope = np.clip(envelope, 0, None) carrier = envelope * np.sin(2 * np.pi * np.cumsum(frequency) / sr) carrier = (carrier / (np.abs(carrier).max() + 1e-9) * 0.9).astype(np.float32) return carrier, float(np.nanmean(voiced)) # Compile pYIN's numba kernels at startup rather than inside the first GPU call. try: prosody_sine(np.sin(np.linspace(0, 2 * np.pi * 220 * 2, SR * 2)).astype(np.float32)) except Exception as _warmup_error: # noqa: BLE001 print(f"pYIN warmup skipped: {_warmup_error}", flush=True) @torch.inference_mode() def carrier_latents(carrier: np.ndarray) -> torch.Tensor: """VAE-encode the stereo-duplicated carrier to [frames, 64] latents at 25 Hz.""" stereo = np.stack([carrier, carrier], 1) chunk = SR * 30 pieces = [] for start in range(0, len(stereo), chunk): segment = stereo[start:start + chunk] if len(segment) < 1920: break tensor = torch.tensor(np.ascontiguousarray(segment.T[None])) pieces.append(vae.encode(tensor)[0].T.float()) if not pieces: raise gr.Error("The hum is too short — please record at least a second of melody.") return torch.cat(pieces, 0) # -------------------------------------------------------------------------------------- # Score helpers # -------------------------------------------------------------------------------------- _VOICE = re.compile(r"^V:\s*(\w+)") _RESTS = re.compile(r"(Z\d*\|)+") _TRAILING = re.compile(r"(V: (Vocal|Ins))|(Z\d*\|)|") def promote_hum_to_vocal(abc: str): """Put the hummed melody on the Vocal staff. SheetSage2 has no words to latch onto in a hum, so it usually files the melody under the `Ins` voice and leaves `Vocal` as whole-bar rests. A hum *is* the vocal line, so when Vocal is rests-only the two staves are swapped before the planner continues the score. """ lines = abc.split("\n") key_index = next((i for i, line in enumerate(lines) if line.startswith("K:")), None) if key_index is None: return abc, False head, body = lines[: key_index + 1], lines[key_index + 1:] blocks, current = [], None for line in body: stripped = line.strip() match = _VOICE.match(stripped) if match: current = {"voice": match.group(1), "header": line, "lines": []} blocks.append(current) elif not stripped or stripped.startswith("%"): blocks.append({"voice": None, "header": line, "lines": []}) current = None elif current is not None: current["lines"].append(line) else: blocks.append({"voice": None, "header": line, "lines": []}) vocal = [b for b in blocks if b["voice"] == "Vocal"] ins = [b for b in blocks if b["voice"] == "Ins"] def rests_only(group): text = "".join(line.strip() for block in group for line in block["lines"]) return bool(text) and _RESTS.fullmatch(text) is not None if not ins or not vocal or not rests_only(vocal) or rests_only(ins): return abc, False for a, b in zip(vocal, ins): a["lines"], b["lines"] = b["lines"], a["lines"] out = list(head) for block in blocks: out.append(block["header"]) out.extend(block["lines"]) return "\n".join(out) + "\n", True def open_score(abc: str) -> str: """Drop trailing rest-only bars / dangling voice markers so the score stays open.""" lines = abc.split("\n") while lines and _TRAILING.fullmatch(lines[-1].strip()): lines.pop() return "\n".join(lines) + "\n" # -------------------------------------------------------------------------------------- # Hum-conditioned flow-matching decoder # -------------------------------------------------------------------------------------- class HumCachedNAR(yue2_nar.CachedNAR): """CachedNAR with the carrier projections added at layers 0 / 7 / 14 / 21.""" def __init__(self, model, chunk, carrier, guidance=1.0, **kwargs): super().__init__(model, chunk, **kwargs) padded = F.pad(carrier, (0, 0, 1, 1))[None].to(self.dtype) self.hum = {layer: hum_proj[slot](padded) for slot, layer in enumerate(INJECT_LAYERS)} self.guidance = float(guidance) self.blank = None if self.guidance != 1.0: zeros = torch.zeros_like(padded) self.blank = {layer: hum_proj[slot](zeros) for slot, layer in enumerate(INJECT_LAYERS)} @torch.inference_mode() def _branch(self, state, raw_t, injection): model = self.model shifted = model._shift_t_value(raw_t, self.device, self.dtype) x = model.vae2llm(F.pad(state, (0, 0, 1, 1))[None].to(self.dtype)) x = x + injection[0] x = x + model.time_embedder(shifted.expand(self.nar_length))[None] x = x + self.pos_emb for index, (layer, (ar_k, ar_v)) in enumerate(zip(model.model.layers, self.cache)): if index and index in injection: x = x + injection[index] q, k, v = layer.nar_self_attn.project_qkv(layer.nar_input_layernorm(x), self.cos, self.sin) k, v = torch.cat((ar_k, k[0])), torch.cat((ar_v, v[0])) h = self._attention(q[0], k, v) x = x + layer.nar_self_attn.o_proj(h.flatten(1)[None]) x = x + layer.nar_mlp(layer.nar_pre_mlp_layernorm(x)) return model.llm2vae(model.model.norm(x))[0, 1:-1].float() @torch.inference_mode() def velocity(self, state, raw_t): conditioned = self._branch(state, raw_t, self.hum) if self.blank is None: return conditioned unconditioned = self._branch(state, raw_t, self.blank) return unconditioned + self.guidance * (conditioned - unconditioned) @torch.inference_mode() def solve(self, steps=32, cancelled=None, on_progress=None): """Reference 32-step midpoint solver, integrated in FP32 as in infer_hum.py.""" state = self.chunk.noise.to(device=self.device) dt = 1.0 / int(steps) for step in range(int(steps)): t = 1.0 - step * dt raw = torch.logit(torch.tensor(t, dtype=torch.float64)).clamp(-20, 20).item() first = self.velocity(state, raw) mid = state - first * (dt / 2) raw_mid = torch.logit(torch.tensor(t - dt / 2, dtype=torch.float64)).clamp(-20, 20).item() state = state - self.velocity(mid, raw_mid) * dt result = state.float().cpu() if not torch.isfinite(result).all(): raise gr.Error("The decoder produced non-finite latents; try another seed.") return result @torch.inference_mode() def hum_synthesize(prefix, codec, condition, seed, steps, guidance): chunks = yue2_nar.song_chunks(prefix, codec, seed, pipe.generation_config.context) pieces, offset = [], 0 for chunk in chunks: frames = len(chunk.noise) engine = HumCachedNAR(model, chunk, condition[offset:offset + frames], guidance=guidance) try: pieces.append(engine.solve(steps)) finally: engine.close() offset += frames del engine return torch.cat(pieces, 0) MELODY_MODES = { "Continue from hum": "continue", "Hum only": "hum_only", "Ignore hum melody": "ignore", } DEFAULT_STYLE = "City Pop, upbeat, danceable, groovy bass, electric guitar, synth, energetic, joyful, neon city night" DEFAULT_LYRICS = """[Intro] [Verse] Neon on the window, humming to the rain Every empty station calls my name again Take the midnight highway, leave the quiet town Turn the radio over, turn the feeling down [Chorus] Hold the night a little longer Every heartbeat getting stronger Sing it back to me Sing it back to me [Outro] Sing it back to me """ def estimate_duration(hum=None, style=None, lyrics=None, melody_mode="Continue from hum", max_seconds=140, hum_influence=1.0, ode_steps=32, seed=831001, *args, **kwargs): """Per-frame cost model fitted to measured runs on this Space's hardware. Measured (frames -> seconds): semantic 0.0095/frame, NAR decode 0.000088/frame/ODE-step/branch, VAE decode 0.00042/frame; pYIN carrier 0.25 s per second of hum. The song length is decided by the planner, so this budgets for the worst case where it fills ``max_seconds``. """ hum_seconds = 0.0 try: path = hum.get("path") if isinstance(hum, dict) else hum if path: info = sf.info(str(path)) hum_seconds = min(MAX_HUM_SECONDS, info.frames / max(1, info.samplerate)) except Exception: hum_seconds = 0.0 hum_seconds = hum_seconds or MAX_HUM_SECONDS steps = max(4, int(ode_steps or 32)) branches = 2.0 if float(hum_influence or 1.0) != 1.0 else 1.0 frames = 25.0 * float(max_seconds or 140) fixed = 6.0 + 0.25 * hum_seconds + 0.15 * hum_seconds # write/overhead + pYIN + transcribe variable = frames * (0.0095 + 0.000088 * steps * branches + 0.00042) return int(min(300, round(1.2 * (fixed + variable)))) @spaces.GPU(duration=estimate_duration) def hum_to_song(hum, style: str = DEFAULT_STYLE, lyrics: str = DEFAULT_LYRICS, melody_mode: str = "Continue from hum", max_seconds: int = 140, hum_influence: float = 1.0, ode_steps: int = 32, seed: int = 831001): """Turn a hummed melody plus a style line and lyrics into a finished song. Args: hum: path to a 10-30 s recording of a hummed melody (voice, not a synth tone). style: comma-separated style / instrumentation tags for the arrangement. lyrics: section-tagged lyrics, e.g. "[Verse] ... [Chorus] ...". melody_mode: "Continue from hum" lets the planner extend your melody into a full song, "Hum only" keeps exactly your melody, "Ignore hum melody" lets the planner write its own tune and uses the hum for phrasing only. max_seconds: hard cap on the rendered song length. hum_influence: classifier-free guidance on the hum channel (1.0 = as trained). ode_steps: flow-matching midpoint steps for the decoder. seed: sampling seed. Returns: The song as MP3, the same song as 24-bit FLAC, the pitch carrier the decoder was conditioned on, the ABC score, and a run summary. """ if not hum: raise gr.Error("Please record or upload a hummed melody first.") style = (style or "").strip() lyrics = (lyrics or "").strip() if not style or not lyrics: raise gr.Error("Please provide both a style line and lyrics.") pipe.device = next(model.parameters()).device mode = MELODY_MODES.get(melody_mode, "continue") seed = int(seed) % (2 ** 31) steps = max(4, int(ode_steps)) guidance = float(hum_influence) timings = {} started = time.perf_counter() # ---- hum -> carrier ------------------------------------------------------------- mark = time.perf_counter() audio = load_audio(hum, SR, True)[: MAX_HUM_SECONDS * SR] if len(audio) < SR: raise gr.Error("That clip is under a second — please hum for 10 to 30 seconds.") carrier, voiced = prosody_sine(audio, SR) if carrier is None or voiced < 0.02: raise gr.Error("No pitch was found in that clip. Hum a melody with your voice, " "close to the microphone.") timings["carrier"] = time.perf_counter() - mark # ---- hum -> ABC ------------------------------------------------------------------ mark = time.perf_counter() swapped = False hum_abc = None if mode != "ignore": try: result = transcriber.transcribe(audio[None], sampling_rate=SR, melody_only=True) hum_abc = result.get("abc") except Exception as error: # noqa: BLE001 raise gr.Error(f"Melody transcription failed: {error}") from error if not hum_abc or not hum_abc.strip(): raise gr.Error("The melody transcriber could not read that clip. Try a clearer, " "steadier hum (it needs a voice, not a synthesised tone).") hum_abc, swapped = promote_hum_to_vocal(hum_abc) timings["transcribe"] = time.perf_counter() - mark # ---- stage 1: plan the score ----------------------------------------------------- mark = time.perf_counter() request = SongRequest(style=style, lyrics=lyrics, cot="melody", seed=seed, id="hum") if mode == "continue": partial = tokenizer.encode(open_score(hum_abc)) prefix = token_prefixes(request, tokenizer) + partial sampling = resolve_sampling(None, pipe.generation_config.abc) ids, timing, truncated = pipe._generate(prefix, sampling, seed, "abc") abc_ids = partial + [int(t) for t in ids if int(t) < EOD] plan = SymbolicPlan(request, tokenizer.decode(abc_ids), abc_ids, token_prefixes(request, tokenizer, abc_ids), timing, truncated) elif mode == "hum_only": request = dataclasses.replace(request, abc=hum_abc) plan = pipe.plan(request=request) else: plan = pipe.plan(request=request) timings["score"] = time.perf_counter() - mark # ---- stage 1b: semantic tokens --------------------------------------------------- mark = time.perf_counter() cap = max(400, int(float(max_seconds) * FRAME_RATE)) headroom = pipe.generation_config.context - len(plan.prefix) - 8 if headroom < 400: raise gr.Error("Those lyrics plus the planned score fill the model's context. " "Please shorten the lyrics.") cap = min(cap, headroom) semantic = pipe.generate_semantic(plan, sampling=resolve_sampling({"max_tokens": cap}, pipe.generation_config.semantic)) codec = list(semantic.tokens) if not codec: raise gr.Error("The planner produced no audio tokens; try another seed.") timings["semantic"] = time.perf_counter() - mark # ---- stage 2: hum-conditioned decode -------------------------------------------- mark = time.perf_counter() latents = carrier_latents(carrier) frames = len(codec) condition = torch.zeros(frames, 64, dtype=torch.float32, device=latents.device) used = min(len(latents), frames) condition[:used] = latents[:used] timings["encode"] = time.perf_counter() - mark mark = time.perf_counter() z = hum_synthesize(plan.prefix, codec, condition, seed, steps, guidance) timings["decoder"] = time.perf_counter() - mark mark = time.perf_counter() with torch.inference_mode(): wave = vae.decode_tiled(z.T[None].contiguous(), core_frames=750, halo_frames=16, output_device="cpu") song = np.nan_to_num(wave[0].float().clamp(-1, 1).T.numpy()).astype(np.float32) timings["vae"] = time.perf_counter() - mark # ---- deliver --------------------------------------------------------------------- out = Path(tempfile.mkdtemp(prefix="humsong_")) flac_path = out / "song.flac" mp3_path = out / "song.mp3" carrier_path = out / "pitch_carrier.mp3" sf.write(flac_path, song, SR, subtype="PCM_24") sf.write(out / "carrier.wav", carrier, SR) for source, target in ((flac_path, mp3_path), (out / "carrier.wav", carrier_path)): subprocess.run(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source), "-codec:a", "libmp3lame", "-b:a", "192k", str(target)], check=True, capture_output=True) total = time.perf_counter() - started seconds = len(song) / SR notes = [ f"**{seconds:.1f} s** of audio in **{total:.0f} s** on ZeroGPU.", f"mode `{mode}` · {steps} ODE steps · hum influence {guidance:g} · seed {seed}", f"hum: {len(audio) / SR:.1f} s, {voiced * 100:.0f}% voiced → " f"{len(latents)} carrier frames over {frames} song frames", "stage timings (s): " + " · ".join(f"{k} {v:.1f}" for k, v in timings.items()), ] if swapped: notes.append("_The transcriber filed the hum under the instrumental staff, so it was " "moved onto the vocal staff before planning._") if semantic.truncated: notes.append("_Hit the length cap — raise **max song length** for a full ending._") return str(mp3_path), str(flac_path), str(carrier_path), plan.abc or "", "\n\n".join(notes) # -------------------------------------------------------------------------------------- # UI # -------------------------------------------------------------------------------------- CSS = """ #col-container { max-width: 1100px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } """ WALTZ_STYLE = ("Indie folk waltz, 3/4, warm female lead vocal, fingerpicked acoustic guitar, " "upright bass, brushed drums, intimate, nostalgic") WALTZ_LYRICS = """[Intro] [Verse] Count the streetlights one two three Slowly turning back to me Every window keeps a song Something I have hummed along [Chorus] Carry me around again Slow and easy like it was then Carry me around again [Outro] Slowly turning back to me """ JAZZ_STYLE = "Jazz-funk, warm lead vocal, Rhodes piano, electric bass, tight drums, brass stabs" JAZZ_LYRICS = """[Intro] [Verse] Morning on the avenue, coffee going cold Somebody is playing something seventy years old Out here on the corner where the traffic learns to swing I have got a little melody and nothing else to bring [Chorus] So play it like you mean it Play it till the daylight ends Play it like you mean it [Outro] Play it till the daylight ends """ EXAMPLES = [ ["examples/hum_02.wav", DEFAULT_STYLE, DEFAULT_LYRICS], ["examples/hum_01.wav", WALTZ_STYLE, WALTZ_LYRICS], ["examples/hum_03.wav", JAZZ_STYLE, JAZZ_LYRICS], ] with gr.Blocks(title="Hum to Song") as demo: with gr.Column(elem_id="col-container"): gr.Markdown( """ # 🎤 Hum to Song Hum a melody for 10–30 seconds, add a style line and lyrics, and get a produced song that keeps your tune, builds a structure around it and carries on long after the hum stops. [`Mothersuperior/YuE2-hum-to-song`](https://huggingface.co/Mothersuperior/YuE2-hum-to-song) on top of [`m-a-p/YuE2-3B`](https://huggingface.co/m-a-p/YuE2-3B): the hum is transcribed with [SheetSage2](https://huggingface.co/m-a-p/SheetSage2), the planner *continues* that open score into a whole song, and a rank-96 adapter feeds your actual pitch contour into the flow-matching decoder. Expect one to two minutes. """ ) with gr.Row(): with gr.Column(): hum = gr.Audio(label="Your hum (10–30 s)", type="filepath", sources=["microphone", "upload"]) style = gr.Textbox(label="Style", value=DEFAULT_STYLE, lines=2, info="Genre, mood and instrumentation tags.") lyrics = gr.Textbox(label="Lyrics", value=DEFAULT_LYRICS, lines=12, info="Use [Intro] / [Verse] / [Chorus] / [Outro] section tags.") melody_mode = gr.Radio( label="Melody mode", choices=list(MELODY_MODES), value="Continue from hum", info="Continue = your melody opens the song and the planner writes the rest.", ) run = gr.Button("Make the song", variant="primary") with gr.Accordion("Advanced", open=False): max_seconds = gr.Slider(60, 200, value=140, step=10, label="Max song length (s)") hum_influence = gr.Slider(1.0, 3.0, value=1.0, step=0.1, label="Hum influence (CFG on the hum channel)", info="1.0 = as trained. Above 1.0 doubles decode time.") ode_steps = gr.Slider(8, 48, value=32, step=4, label="Decoder ODE steps") seed = gr.Slider(0, 2 ** 31 - 1, value=831001, step=1, label="Seed") with gr.Column(): song_out = gr.Audio(label="Song", type="filepath") info_out = gr.Markdown() flac_out = gr.File(label="24-bit FLAC") carrier_out = gr.Audio(label="Pitch carrier — all the decoder sees of your hum", type="filepath") score_out = gr.Code(label="Score (ABC)", lines=18) gr.Examples( examples=EXAMPLES, inputs=[hum, style, lyrics], outputs=[song_out, flac_out, carrier_out, score_out, info_out], fn=hum_to_song, cache_examples=True, cache_mode="lazy", label="Example hums (CHAD, CC BY-NC 4.0)", ) gr.Markdown( "Weights derive from YuE2-3B and are **CC BY-NC 4.0 — non-commercial use only**. " "Example hums come from the " "[CHAD hummings subset](https://huggingface.co/datasets/amanteur/CHAD_hummings) " "(Amatov et al., ISMIR 2023), CC BY-NC 4.0." ) run.click( fn=hum_to_song, inputs=[hum, style, lyrics, melody_mode, max_seconds, hum_influence, ode_steps, seed], outputs=[song_out, flac_out, carrier_out, score_out, info_out], api_name="hum_to_song", ) demo.queue(max_size=12).launch( theme=gr.themes.Citrus(), css=CSS, mcp_server=True, show_error=True )