diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index 16463fa..6791c3c 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -74,7 +74,8 @@ Titelgenerierung und Kontextkompression in Hermes. | Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg | |---|---|---|---|---|---| -| 08.09.2026 | `ACE-Step/acestep-v15-xl-sft` mit `acestep-5Hz-lm-1.7B`, offizielles ACE-Step-1.5-Image `sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567` | 30 s Instrumental, Thinking/LM aktiv, Batch 1, RTX 5080 16 GiB, automatischer CPU-Offload und INT8 Weight-only DiT | erfolgreich in 15,39 s: LM 8,00 s, DiT 7,39 s, MP3 0,82 s; PyTorch meldete maximal 9,38 GiB CUDA-Allokation; kein OOM/CUDA-Fehler | **Beta-Test bestanden**; Klangabnahme und Hermes-/Router-Integration noch offen | Athena: `/data/music/acestep/batch_1788873234/`; [SPECIALIZED_MODEL_ROADMAP.md](SPECIALIZED_MODEL_ROADMAP.md) | +| 08.–10.09.2026 | `ACE-Step/acestep-v15-xl-sft` mit `acestep-5Hz-lm-1.7B`, offizielles ACE-Step-1.5-Image `sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567` | Mehrere Instrumentaltests bis 244 s; abschließender Kontrolllauf mit geladenem 1.7B-Planer, `thinking=True`, XL-SFT 4B, 80 Schritten, Guidance 8 und Shift 3 | technisch vollständig und schnell, aber wiederholt nur Geräusche/Krach oder musikalisch chaotische Ergebnisse; der letzte Lauf schließt einen bloß fehlenden Planer als Ursache aus | **qualitativ verworfen**; Container und Daten zunächst nur für den direkten HeartMuLa-A/B-Rückweg erhalten | Athena: `/data/music/acestep/`; [SPECIALIZED_MODEL_ROADMAP.md](SPECIALIZED_MODEL_ROADMAP.md) | +| 10.09.2026 | `HeartMuLa/HeartMuLa-oss-3B-happy-new-year` mit `HeartMuLa/HeartCodec-oss-20260123`, heartlib `3783bdb8441f2c298b1e64c8651173aac200361c` | Isolierter 20-s-Instrumentaltest mit offiziellen Standardwerten; HeartMuLa BF16 auf RTX 5080, HeartCodec FP32 auf RTX 3060 | Modell und Codec laden stabil; 20,08-s-Stereoausgabe erfolgreich in rund 27 s erzeugt, 48 kHz/24-Bit-PCM. Referenzaudio wird upstream noch nicht unterstützt, Instrumentalsteuerung ist experimentell | **technischer Test bestanden, Hörabnahme offen** | [heartmula-3b](../experiments/heartmula-3b/README.md); Athena: `/data/music/heartmula/heartmula-1789064239-b3f2ca75.wav` | ## Audio-Trennung diff --git a/experiments/heartmula-3b/Dockerfile b/experiments/heartmula-3b/Dockerfile new file mode 100644 index 0000000..38be00a --- /dev/null +++ b/experiments/heartmula-3b/Dockerfile @@ -0,0 +1,22 @@ +FROM pytorch/pytorch:2.9.1-cuda12.8-cudnn9-runtime + +ARG HEARTLIB_COMMIT=3783bdb8441f2c298b1e64c8651173aac200361c + +RUN apt-get update \ + && apt-get install -y --no-install-recommends git ffmpeg curl \ + && rm -rf /var/lib/apt/lists/* + +RUN git clone https://github.com/HeartMuLa/heartlib.git /opt/heartlib \ + && git -C /opt/heartlib checkout "${HEARTLIB_COMMIT}" \ + && pip install --no-cache-dir -e /opt/heartlib \ + && pip install --no-cache-dir gradio==5.49.1 uvicorn==0.35.0 + +COPY app.py /opt/app/app.py + +WORKDIR /opt/app +ENV PYTHONUNBUFFERED=1 \ + HEARTMULA_MODEL_PATH=/models/ckpt \ + HEARTMULA_OUTPUT_DIR=/output + +EXPOSE 7860 +CMD ["python", "/opt/app/app.py"] diff --git a/experiments/heartmula-3b/README.md b/experiments/heartmula-3b/README.md new file mode 100644 index 0000000..133ad1b --- /dev/null +++ b/experiments/heartmula-3b/README.md @@ -0,0 +1,49 @@ +# HeartMuLa 3B quality test + +Reversible A/B test of the public +`HeartMuLa/HeartMuLa-oss-3B-happy-new-year` model with +`HeartMuLa/HeartCodec-oss-20260123`. + +The model and codec are deliberately split across Athena's GPUs: + +- HeartMuLa BF16: RTX 5080 (`cuda:0` inside the container) +- HeartCodec FP32: RTX 3060 (`cuda:1` inside the container) + +The public model does not yet support reference-audio conditioning. Pure +instrumental generation is exposed as an explicitly experimental option. +Generated files are lossless 48-kHz WAV files under `/data/music/heartmula`. + +The container intentionally has no `com.mike-ai.music-worker` label while it +is being evaluated. This prevents it from conflicting with the production +profile controller's single ACE-Step worker. + +## Checkpoints + +The expected persistent layout on Athena is: + +```text +/data/models/heartmula/ckpt/ +├── HeartMuLa-oss-3B/ +├── HeartCodec-oss/ +├── gen_config.json +└── tokenizer.json +``` + +## Test operation + +Only after the current music worker has been stopped: + +```sh +cd /opt/mike-ai/heartmula-3b +docker compose up -d heartmula-test +docker compose logs -f heartmula-test +``` + +The temporary test UI is then available on port `7863`. Stop it with: + +```sh +docker compose down +``` + +Do not remove ACE-Step or change the dashboard/profile controller until the +A/B test has been accepted. diff --git a/experiments/heartmula-3b/app.py b/experiments/heartmula-3b/app.py new file mode 100644 index 0000000..a34f800 --- /dev/null +++ b/experiments/heartmula-3b/app.py @@ -0,0 +1,148 @@ +from __future__ import annotations + +import os +import threading +import time +import uuid +from pathlib import Path + +import gradio as gr +import soundfile as sf +import torch +import uvicorn +from fastapi import FastAPI +from heartlib import HeartMuLaGenPipeline + + +MODEL_PATH = os.environ.get("HEARTMULA_MODEL_PATH", "/models/ckpt") +OUTPUT_DIR = Path(os.environ.get("HEARTMULA_OUTPUT_DIR", "/output")) +MULA_DEVICE = os.environ.get("HEARTMULA_MULA_DEVICE", "cuda:0") +CODEC_DEVICE = os.environ.get("HEARTMULA_CODEC_DEVICE", "cuda:1") + +OUTPUT_DIR.mkdir(parents=True, exist_ok=True) + +pipeline: HeartMuLaGenPipeline | None = None +load_error = "" +generation_lock = threading.Lock() + + +class AthenaHeartMuLaPipeline(HeartMuLaGenPipeline): + """Write 24-bit WAV directly; torch 2.9 otherwise requires TorchCodec.""" + + def postprocess(self, model_outputs, save_path: str): + frames = model_outputs["frames"].to(self.codec_device) + wav = self.codec.detokenize(frames) + self._unload() + audio = wav.to(torch.float32).cpu().numpy().T + sf.write(save_path, audio, 48_000, format="WAV", subtype="PCM_24") + + +def load_pipeline() -> None: + global pipeline, load_error + try: + pipeline = AthenaHeartMuLaPipeline.from_pretrained( + MODEL_PATH, + device={ + "mula": torch.device(MULA_DEVICE), + "codec": torch.device(CODEC_DEVICE), + }, + dtype={"mula": torch.bfloat16, "codec": torch.float32}, + version="3B", + lazy_load=False, + ) + except Exception as exc: + load_error = f"{type(exc).__name__}: {exc}" + raise + + +def generate( + tags: str, + lyrics: str, + instrumental: bool, + duration_seconds: int, + topk: int, + temperature: float, + cfg_scale: float, +): + if pipeline is None: + raise gr.Error(f"Modell ist nicht bereit. {load_error}".strip()) + tags = ",".join(part.strip() for part in tags.split(",") if part.strip()) + if not tags: + raise gr.Error("Bitte mindestens ein Musik-Tag angeben.") + if instrumental: + effective_lyrics = "[Instrumental]" + else: + effective_lyrics = lyrics.strip() + if not effective_lyrics: + raise gr.Error("Für ein Lied mit Gesang fehlt der Liedtext.") + + target = OUTPUT_DIR / f"heartmula-{int(time.time())}-{uuid.uuid4().hex[:8]}.wav" + with generation_lock, torch.inference_mode(): + pipeline( + {"lyrics": effective_lyrics, "tags": tags}, + max_audio_length_ms=int(duration_seconds) * 1000, + save_path=str(target), + topk=int(topk), + temperature=float(temperature), + cfg_scale=float(cfg_scale), + ) + return str(target), ( + f"Fertig: {target.name} · WAV 48 kHz. " + "Instrumental ist bei der öffentlichen 3B-Version experimentell." + ) + + +with gr.Blocks(title="HeartMuLa 3B – Athena Test") as demo: + gr.Markdown( + "# HeartMuLa 3B – Athena Test\n" + "Offizielles öffentliches 3B-Modell. HeartMuLa läuft auf der RTX 5080, " + "HeartCodec verlustarm in FP32 auf der RTX 3060. " + "Referenzaudio wird von dieser Version noch nicht unterstützt." + ) + tags = gr.Textbox( + label="Stil und Instrumente (Komma-getrennte Tags)", + value="synthwave,retrowave,1980s,melodic,nostalgic,arpeggiated synthesizer,drum machine,atmospheric,cinematic", + lines=3, + ) + lyrics = gr.Textbox( + label="Liedtext mit Abschnitten wie [Verse], [Chorus], [Bridge]", + lines=14, + placeholder="[Verse]\n...\n\n[Chorus]\n...", + ) + instrumental = gr.Checkbox(label="Instrumental (experimentell)", value=True) + duration = gr.Slider(20, 240, value=60, step=5, label="Maximale Dauer in Sekunden") + with gr.Accordion("Sampling – offizielle Standardwerte", open=False): + topk = gr.Slider(1, 200, value=50, step=1, label="Top-k") + temperature = gr.Slider(0.1, 2.0, value=1.0, step=0.05, label="Temperatur") + cfg_scale = gr.Slider(1.0, 4.0, value=1.5, step=0.1, label="CFG") + run = gr.Button("Musik erzeugen", variant="primary") + audio = gr.Audio(label="Ergebnis", type="filepath") + status = gr.Textbox(label="Status", interactive=False) + run.click( + generate, + inputs=[tags, lyrics, instrumental, duration, topk, temperature, cfg_scale], + outputs=[audio, status], + ) + +demo.queue(default_concurrency_limit=1, max_size=4) + +api = FastAPI() + + +@api.get("/healthz") +def healthz(): + return { + "ready": pipeline is not None, + "error": load_error or None, + "model": "HeartMuLa-oss-3B-happy-new-year", + "codec": "HeartCodec-oss-20260123", + "mula_device": MULA_DEVICE, + "codec_device": CODEC_DEVICE, + } + + +api = gr.mount_gradio_app(api, demo, path="/") + +if __name__ == "__main__": + load_pipeline() + uvicorn.run(api, host="0.0.0.0", port=7860) diff --git a/experiments/heartmula-3b/compose.yaml b/experiments/heartmula-3b/compose.yaml new file mode 100644 index 0000000..de72ee9 --- /dev/null +++ b/experiments/heartmula-3b/compose.yaml @@ -0,0 +1,32 @@ +services: + heartmula-test: + build: + context: . + args: + HEARTLIB_COMMIT: 3783bdb8441f2c298b1e64c8651173aac200361c + image: mike-ai/heartmula:3b-3783bdb + container_name: mike-ai-heartmula-test + restart: "no" + environment: + NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b" + NVIDIA_DRIVER_CAPABILITIES: compute,utility + HEARTMULA_MULA_DEVICE: cuda:0 + HEARTMULA_CODEC_DEVICE: cuda:1 + volumes: + - /data/models/heartmula:/models:ro + - /data/music/heartmula:/output + ports: + - "7863:7860" + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: all + capabilities: [gpu] + healthcheck: + test: ["CMD", "curl", "-fsS", "http://127.0.0.1:7860/healthz"] + interval: 15s + timeout: 5s + retries: 20 + start_period: 180s