Prepare isolated YuE2 music evaluation

This commit is contained in:
Mikei386
2026-09-10 21:25:59 +02:00
parent 6bd6c48a95
commit 1904f2104f
9 changed files with 102 additions and 263 deletions
-22
View File
@@ -1,22 +0,0 @@
FROM pytorch/pytorch:2.9.1-cuda12.8-cudnn9-runtime
ARG HEARTLIB_COMMIT=3783bdb8441f2c298b1e64c8651173aac200361c
RUN apt-get update \
&& apt-get install -y --no-install-recommends git ffmpeg curl \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/HeartMuLa/heartlib.git /opt/heartlib \
&& git -C /opt/heartlib checkout "${HEARTLIB_COMMIT}" \
&& pip install --no-cache-dir -e /opt/heartlib \
&& pip install --no-cache-dir gradio==5.49.1 uvicorn==0.35.0
COPY app.py /opt/app/app.py
WORKDIR /opt/app
ENV PYTHONUNBUFFERED=1 \
HEARTMULA_MODEL_PATH=/models/ckpt \
HEARTMULA_OUTPUT_DIR=/output
EXPOSE 7860
CMD ["python", "/opt/app/app.py"]
-52
View File
@@ -1,52 +0,0 @@
# HeartMuLa 3B quality test
Reversible A/B test of the public
`HeartMuLa/HeartMuLa-oss-3B-happy-new-year` model with
`HeartMuLa/HeartCodec-oss-20260123`.
The model and codec are deliberately split across Athena's GPUs:
- HeartMuLa BF16: RTX 5080 (`cuda:0` inside the container)
- HeartCodec FP32: RTX 3060 (`cuda:1` inside the container)
The public model does not yet support reference-audio conditioning. Pure
instrumental generation is exposed as an explicitly experimental option.
Generated files are lossless 48-kHz WAV files under `/data/music/heartmula`.
The container intentionally has no `com.mike-ai.music-worker` label while it
is being evaluated. This prevents it from conflicting with the production
profile controller's single ACE-Step worker.
## Checkpoints
The expected persistent layout on Athena is:
```text
/data/models/heartmula/ckpt/
├── HeartMuLa-oss-3B/
├── HeartCodec-oss/
├── gen_config.json
└── tokenizer.json
```
## Test operation
Only after the current music worker has been stopped:
```sh
cd /opt/mike-ai/heartmula-3b
docker compose up -d heartmula-test
docker compose logs -f heartmula-test
```
The temporary test UI listens locally on port `7863`. During the listening
test it is proxied through Athena's existing WireGuard gateway; permanent
dashboard/gateway integration is deliberately deferred until the audio has
been accepted. Stop it with:
```sh
docker compose down
```
Do not remove ACE-Step or change the dashboard/profile controller until the
A/B test has been accepted.
-148
View File
@@ -1,148 +0,0 @@
from __future__ import annotations
import os
import threading
import time
import uuid
from pathlib import Path
import gradio as gr
import soundfile as sf
import torch
import uvicorn
from fastapi import FastAPI
from heartlib import HeartMuLaGenPipeline
MODEL_PATH = os.environ.get("HEARTMULA_MODEL_PATH", "/models/ckpt")
OUTPUT_DIR = Path(os.environ.get("HEARTMULA_OUTPUT_DIR", "/output"))
MULA_DEVICE = os.environ.get("HEARTMULA_MULA_DEVICE", "cuda:0")
CODEC_DEVICE = os.environ.get("HEARTMULA_CODEC_DEVICE", "cuda:1")
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
pipeline: HeartMuLaGenPipeline | None = None
load_error = ""
generation_lock = threading.Lock()
class AthenaHeartMuLaPipeline(HeartMuLaGenPipeline):
"""Write 24-bit WAV directly; torch 2.9 otherwise requires TorchCodec."""
def postprocess(self, model_outputs, save_path: str):
frames = model_outputs["frames"].to(self.codec_device)
wav = self.codec.detokenize(frames)
self._unload()
audio = wav.to(torch.float32).cpu().numpy().T
sf.write(save_path, audio, 48_000, format="WAV", subtype="PCM_24")
def load_pipeline() -> None:
global pipeline, load_error
try:
pipeline = AthenaHeartMuLaPipeline.from_pretrained(
MODEL_PATH,
device={
"mula": torch.device(MULA_DEVICE),
"codec": torch.device(CODEC_DEVICE),
},
dtype={"mula": torch.bfloat16, "codec": torch.float32},
version="3B",
lazy_load=False,
)
except Exception as exc:
load_error = f"{type(exc).__name__}: {exc}"
raise
def generate(
tags: str,
lyrics: str,
instrumental: bool,
duration_seconds: int,
topk: int,
temperature: float,
cfg_scale: float,
):
if pipeline is None:
raise gr.Error(f"Modell ist nicht bereit. {load_error}".strip())
tags = ",".join(part.strip() for part in tags.split(",") if part.strip())
if not tags:
raise gr.Error("Bitte mindestens ein Musik-Tag angeben.")
if instrumental:
effective_lyrics = "[Instrumental]"
else:
effective_lyrics = lyrics.strip()
if not effective_lyrics:
raise gr.Error("Für ein Lied mit Gesang fehlt der Liedtext.")
target = OUTPUT_DIR / f"heartmula-{int(time.time())}-{uuid.uuid4().hex[:8]}.wav"
with generation_lock, torch.inference_mode():
pipeline(
{"lyrics": effective_lyrics, "tags": tags},
max_audio_length_ms=int(duration_seconds) * 1000,
save_path=str(target),
topk=int(topk),
temperature=float(temperature),
cfg_scale=float(cfg_scale),
)
return str(target), (
f"Fertig: {target.name} · WAV 48 kHz. "
"Instrumental ist bei der öffentlichen 3B-Version experimentell."
)
with gr.Blocks(title="HeartMuLa 3B – Athena Test") as demo:
gr.Markdown(
"# HeartMuLa 3B – Athena Test\n"
"Offizielles öffentliches 3B-Modell. HeartMuLa läuft auf der RTX 5080, "
"HeartCodec verlustarm in FP32 auf der RTX 3060. "
"Referenzaudio wird von dieser Version noch nicht unterstützt."
)
tags = gr.Textbox(
label="Stil und Instrumente (Komma-getrennte Tags)",
value="synthwave,retrowave,1980s,melodic,nostalgic,arpeggiated synthesizer,drum machine,atmospheric,cinematic",
lines=3,
)
lyrics = gr.Textbox(
label="Liedtext mit Abschnitten wie [Verse], [Chorus], [Bridge]",
lines=14,
placeholder="[Verse]\n...\n\n[Chorus]\n...",
)
instrumental = gr.Checkbox(label="Instrumental (experimentell)", value=True)
duration = gr.Slider(20, 240, value=60, step=5, label="Maximale Dauer in Sekunden")
with gr.Accordion("Sampling – offizielle Standardwerte", open=False):
topk = gr.Slider(1, 200, value=50, step=1, label="Top-k")
temperature = gr.Slider(0.1, 2.0, value=1.0, step=0.05, label="Temperatur")
cfg_scale = gr.Slider(1.0, 4.0, value=1.5, step=0.1, label="CFG")
run = gr.Button("Musik erzeugen", variant="primary")
audio = gr.Audio(label="Ergebnis", type="filepath")
status = gr.Textbox(label="Status", interactive=False)
run.click(
generate,
inputs=[tags, lyrics, instrumental, duration, topk, temperature, cfg_scale],
outputs=[audio, status],
)
demo.queue(default_concurrency_limit=1, max_size=4)
api = FastAPI()
@api.get("/healthz")
def healthz():
return {
"ready": pipeline is not None,
"error": load_error or None,
"model": "HeartMuLa-oss-3B-happy-new-year",
"codec": "HeartCodec-oss-20260123",
"mula_device": MULA_DEVICE,
"codec_device": CODEC_DEVICE,
}
api = gr.mount_gradio_app(api, demo, path="/")
if __name__ == "__main__":
load_pipeline()
uvicorn.run(api, host="0.0.0.0", port=7860)
-39
View File
@@ -1,39 +0,0 @@
services:
heartmula-test:
build:
context: .
args:
HEARTLIB_COMMIT: 3783bdb8441f2c298b1e64c8651173aac200361c
image: mike-ai/heartmula:3b-3783bdb
container_name: mike-ai-heartmula-test
restart: "no"
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HEARTMULA_MULA_DEVICE: cuda:0
HEARTMULA_CODEC_DEVICE: cuda:1
volumes:
- /data/models/heartmula:/models:ro
- /data/music/heartmula:/output
ports:
- "127.0.0.1:7863:7860"
networks:
- frontend
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:7860/healthz"]
interval: 15s
timeout: 5s
retries: 20
start_period: 180s
networks:
frontend:
external: true
name: mike-ai_frontend
+17
View File
@@ -0,0 +1,17 @@
FROM python:3.12-slim-bookworm
ARG YUE2_COMMIT=9c6c4b349be978b06a9d0d958471a07a6cdeff4d
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates git libsndfile1 \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/multimodal-art-projection/YuE.git /opt/yue2 \
&& git -C /opt/yue2 checkout "${YUE2_COMMIT}" \
&& python -m pip install --no-cache-dir /opt/yue2
WORKDIR /workspace
ENV PYTHONUNBUFFERED=1 \
YUE2_KIT=/workspace
ENTRYPOINT ["yue2"]
+52
View File
@@ -0,0 +1,52 @@
# YuE2 3B isolated quality test
Prepared, non-starting evaluation of `m-a-p/YuE2-3B` with the standard
`m-a-p/YuE2-Vae` listening decoder. The source is pinned to the official
`yue2-v0.1.6` commit `9c6c4b349be978b06a9d0d958471a07a6cdeff4d`.
Preparation on Athena is complete. The model and VAE files were checked
against their published `weights_manifest.json` SHA-256 values. The Docker
image is built, but no YuE2 container has been created or started.
## Safety and isolation
- This experiment is not part of the profile controller or dashboard.
- The Compose service uses the `manual` profile, has no restart policy and
cannot start through an ordinary `docker compose up`.
- Only the RTX 5080 is exposed to the container.
- Building and downloading do not load the model or use a GPU.
- Do not start it while another Athena GPU job is active.
## Persistent files
```text
/data/models/yue2/
├── YuE2-3B/
└── YuE2-Vae/
/data/music/yue2/
```
The initial control request is a true empty-lyrics instrumental request. No
invented `[Instrumental]` lyrics marker is used.
## Manual test (only after GPU availability was checked)
From `/opt/mike-ai/yue2-3b` on Athena:
```sh
docker compose --profile manual run --rm yue2-test generate \
--offline \
--device cuda:0 \
--budget 16 \
--request /workspace/requests/instrumental-synthwave.json \
--output /workspace/runs
```
Start with the official unquantized BF16 path. If and only if this fails from
VRAM pressure, repeat with `--quantization fp8 --offload-ar`; keep the outputs
separate because that is a different inference configuration.
YuE2 is newly released and officially specifies a 24-GB BF16 GPU. Readiness of
this image and the downloaded weights is not evidence that the 16-GB RTX 5080
run will fit or that its audio quality is acceptable.
+23
View File
@@ -0,0 +1,23 @@
services:
yue2-test:
profiles: ["manual"]
build:
context: .
args:
YUE2_COMMIT: 9c6c4b349be978b06a9d0d958471a07a6cdeff4d
image: mike-ai/yue2:3b-0.1.6
restart: "no"
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
volumes:
- /data/models/yue2:/workspace/models:ro
- /data/music/yue2:/workspace/runs
- ./requests:/workspace/requests:ro
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"]
capabilities: [gpu]
@@ -0,0 +1,7 @@
{
"id": "instrumental_synthwave_control",
"style": "Instrumental, synthwave, synth-pop, energetic, memorable lead melody, arpeggiated synthesizer, analog synthesizer, drum machine, driving bass, 118 BPM, no vocals",
"lyrics": "",
"cot": "full",
"seed": 831001
}