Add isolated HeartMuLa 3B quality test

This commit is contained in:
Mikei386
2026-09-10 20:20:51 +02:00
parent 7c95dab324
commit fe2a93eeb1
5 changed files with 253 additions and 1 deletions
+2 -1
View File
@@ -74,7 +74,8 @@ Titelgenerierung und Kontextkompression in Hermes.
| Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg | | Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---|---|---|---| |---|---|---|---|---|---|
| 08.09.2026 | `ACE-Step/acestep-v15-xl-sft` mit `acestep-5Hz-lm-1.7B`, offizielles ACE-Step-1.5-Image `sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567` | 30 s Instrumental, Thinking/LM aktiv, Batch 1, RTX 5080 16 GiB, automatischer CPU-Offload und INT8 Weight-only DiT | erfolgreich in 15,39 s: LM 8,00 s, DiT 7,39 s, MP3 0,82 s; PyTorch meldete maximal 9,38 GiB CUDA-Allokation; kein OOM/CUDA-Fehler | **Beta-Test bestanden**; Klangabnahme und Hermes-/Router-Integration noch offen | Athena: `/data/music/acestep/batch_1788873234/`; [SPECIALIZED_MODEL_ROADMAP.md](SPECIALIZED_MODEL_ROADMAP.md) | | 08.–10.09.2026 | `ACE-Step/acestep-v15-xl-sft` mit `acestep-5Hz-lm-1.7B`, offizielles ACE-Step-1.5-Image `sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567` | Mehrere Instrumentaltests bis 244 s; abschließender Kontrolllauf mit geladenem 1.7B-Planer, `thinking=True`, XL-SFT 4B, 80 Schritten, Guidance 8 und Shift 3 | technisch vollständig und schnell, aber wiederholt nur Geräusche/Krach oder musikalisch chaotische Ergebnisse; der letzte Lauf schließt einen bloß fehlenden Planer als Ursache aus | **qualitativ verworfen**; Container und Daten zunächst nur für den direkten HeartMuLa-A/B-Rückweg erhalten | Athena: `/data/music/acestep/`; [SPECIALIZED_MODEL_ROADMAP.md](SPECIALIZED_MODEL_ROADMAP.md) |
| 10.09.2026 | `HeartMuLa/HeartMuLa-oss-3B-happy-new-year` mit `HeartMuLa/HeartCodec-oss-20260123`, heartlib `3783bdb8441f2c298b1e64c8651173aac200361c` | Isolierter 20-s-Instrumentaltest mit offiziellen Standardwerten; HeartMuLa BF16 auf RTX 5080, HeartCodec FP32 auf RTX 3060 | Modell und Codec laden stabil; 20,08-s-Stereoausgabe erfolgreich in rund 27 s erzeugt, 48 kHz/24-Bit-PCM. Referenzaudio wird upstream noch nicht unterstützt, Instrumentalsteuerung ist experimentell | **technischer Test bestanden, Hörabnahme offen** | [heartmula-3b](../experiments/heartmula-3b/README.md); Athena: `/data/music/heartmula/heartmula-1789064239-b3f2ca75.wav` |
## Audio-Trennung ## Audio-Trennung
+22
View File
@@ -0,0 +1,22 @@
FROM pytorch/pytorch:2.9.1-cuda12.8-cudnn9-runtime
ARG HEARTLIB_COMMIT=3783bdb8441f2c298b1e64c8651173aac200361c
RUN apt-get update \
&& apt-get install -y --no-install-recommends git ffmpeg curl \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/HeartMuLa/heartlib.git /opt/heartlib \
&& git -C /opt/heartlib checkout "${HEARTLIB_COMMIT}" \
&& pip install --no-cache-dir -e /opt/heartlib \
&& pip install --no-cache-dir gradio==5.49.1 uvicorn==0.35.0
COPY app.py /opt/app/app.py
WORKDIR /opt/app
ENV PYTHONUNBUFFERED=1 \
HEARTMULA_MODEL_PATH=/models/ckpt \
HEARTMULA_OUTPUT_DIR=/output
EXPOSE 7860
CMD ["python", "/opt/app/app.py"]
+49
View File
@@ -0,0 +1,49 @@
# HeartMuLa 3B quality test
Reversible A/B test of the public
`HeartMuLa/HeartMuLa-oss-3B-happy-new-year` model with
`HeartMuLa/HeartCodec-oss-20260123`.
The model and codec are deliberately split across Athena's GPUs:
- HeartMuLa BF16: RTX 5080 (`cuda:0` inside the container)
- HeartCodec FP32: RTX 3060 (`cuda:1` inside the container)
The public model does not yet support reference-audio conditioning. Pure
instrumental generation is exposed as an explicitly experimental option.
Generated files are lossless 48-kHz WAV files under `/data/music/heartmula`.
The container intentionally has no `com.mike-ai.music-worker` label while it
is being evaluated. This prevents it from conflicting with the production
profile controller's single ACE-Step worker.
## Checkpoints
The expected persistent layout on Athena is:
```text
/data/models/heartmula/ckpt/
├── HeartMuLa-oss-3B/
├── HeartCodec-oss/
├── gen_config.json
└── tokenizer.json
```
## Test operation
Only after the current music worker has been stopped:
```sh
cd /opt/mike-ai/heartmula-3b
docker compose up -d heartmula-test
docker compose logs -f heartmula-test
```
The temporary test UI is then available on port `7863`. Stop it with:
```sh
docker compose down
```
Do not remove ACE-Step or change the dashboard/profile controller until the
A/B test has been accepted.
+148
View File
@@ -0,0 +1,148 @@
from __future__ import annotations
import os
import threading
import time
import uuid
from pathlib import Path
import gradio as gr
import soundfile as sf
import torch
import uvicorn
from fastapi import FastAPI
from heartlib import HeartMuLaGenPipeline
MODEL_PATH = os.environ.get("HEARTMULA_MODEL_PATH", "/models/ckpt")
OUTPUT_DIR = Path(os.environ.get("HEARTMULA_OUTPUT_DIR", "/output"))
MULA_DEVICE = os.environ.get("HEARTMULA_MULA_DEVICE", "cuda:0")
CODEC_DEVICE = os.environ.get("HEARTMULA_CODEC_DEVICE", "cuda:1")
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
pipeline: HeartMuLaGenPipeline | None = None
load_error = ""
generation_lock = threading.Lock()
class AthenaHeartMuLaPipeline(HeartMuLaGenPipeline):
"""Write 24-bit WAV directly; torch 2.9 otherwise requires TorchCodec."""
def postprocess(self, model_outputs, save_path: str):
frames = model_outputs["frames"].to(self.codec_device)
wav = self.codec.detokenize(frames)
self._unload()
audio = wav.to(torch.float32).cpu().numpy().T
sf.write(save_path, audio, 48_000, format="WAV", subtype="PCM_24")
def load_pipeline() -> None:
global pipeline, load_error
try:
pipeline = AthenaHeartMuLaPipeline.from_pretrained(
MODEL_PATH,
device={
"mula": torch.device(MULA_DEVICE),
"codec": torch.device(CODEC_DEVICE),
},
dtype={"mula": torch.bfloat16, "codec": torch.float32},
version="3B",
lazy_load=False,
)
except Exception as exc:
load_error = f"{type(exc).__name__}: {exc}"
raise
def generate(
tags: str,
lyrics: str,
instrumental: bool,
duration_seconds: int,
topk: int,
temperature: float,
cfg_scale: float,
):
if pipeline is None:
raise gr.Error(f"Modell ist nicht bereit. {load_error}".strip())
tags = ",".join(part.strip() for part in tags.split(",") if part.strip())
if not tags:
raise gr.Error("Bitte mindestens ein Musik-Tag angeben.")
if instrumental:
effective_lyrics = "[Instrumental]"
else:
effective_lyrics = lyrics.strip()
if not effective_lyrics:
raise gr.Error("Für ein Lied mit Gesang fehlt der Liedtext.")
target = OUTPUT_DIR / f"heartmula-{int(time.time())}-{uuid.uuid4().hex[:8]}.wav"
with generation_lock, torch.inference_mode():
pipeline(
{"lyrics": effective_lyrics, "tags": tags},
max_audio_length_ms=int(duration_seconds) * 1000,
save_path=str(target),
topk=int(topk),
temperature=float(temperature),
cfg_scale=float(cfg_scale),
)
return str(target), (
f"Fertig: {target.name} · WAV 48 kHz. "
"Instrumental ist bei der öffentlichen 3B-Version experimentell."
)
with gr.Blocks(title="HeartMuLa 3B – Athena Test") as demo:
gr.Markdown(
"# HeartMuLa 3B – Athena Test\n"
"Offizielles öffentliches 3B-Modell. HeartMuLa läuft auf der RTX 5080, "
"HeartCodec verlustarm in FP32 auf der RTX 3060. "
"Referenzaudio wird von dieser Version noch nicht unterstützt."
)
tags = gr.Textbox(
label="Stil und Instrumente (Komma-getrennte Tags)",
value="synthwave,retrowave,1980s,melodic,nostalgic,arpeggiated synthesizer,drum machine,atmospheric,cinematic",
lines=3,
)
lyrics = gr.Textbox(
label="Liedtext mit Abschnitten wie [Verse], [Chorus], [Bridge]",
lines=14,
placeholder="[Verse]\n...\n\n[Chorus]\n...",
)
instrumental = gr.Checkbox(label="Instrumental (experimentell)", value=True)
duration = gr.Slider(20, 240, value=60, step=5, label="Maximale Dauer in Sekunden")
with gr.Accordion("Sampling – offizielle Standardwerte", open=False):
topk = gr.Slider(1, 200, value=50, step=1, label="Top-k")
temperature = gr.Slider(0.1, 2.0, value=1.0, step=0.05, label="Temperatur")
cfg_scale = gr.Slider(1.0, 4.0, value=1.5, step=0.1, label="CFG")
run = gr.Button("Musik erzeugen", variant="primary")
audio = gr.Audio(label="Ergebnis", type="filepath")
status = gr.Textbox(label="Status", interactive=False)
run.click(
generate,
inputs=[tags, lyrics, instrumental, duration, topk, temperature, cfg_scale],
outputs=[audio, status],
)
demo.queue(default_concurrency_limit=1, max_size=4)
api = FastAPI()
@api.get("/healthz")
def healthz():
return {
"ready": pipeline is not None,
"error": load_error or None,
"model": "HeartMuLa-oss-3B-happy-new-year",
"codec": "HeartCodec-oss-20260123",
"mula_device": MULA_DEVICE,
"codec_device": CODEC_DEVICE,
}
api = gr.mount_gradio_app(api, demo, path="/")
if __name__ == "__main__":
load_pipeline()
uvicorn.run(api, host="0.0.0.0", port=7860)
+32
View File
@@ -0,0 +1,32 @@
services:
heartmula-test:
build:
context: .
args:
HEARTLIB_COMMIT: 3783bdb8441f2c298b1e64c8651173aac200361c
image: mike-ai/heartmula:3b-3783bdb
container_name: mike-ai-heartmula-test
restart: "no"
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HEARTMULA_MULA_DEVICE: cuda:0
HEARTMULA_CODEC_DEVICE: cuda:1
volumes:
- /data/models/heartmula:/models:ro
- /data/music/heartmula:/output
ports:
- "7863:7860"
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:7860/healthz"]
interval: 15s
timeout: 5s
retries: 20
start_period: 180s