diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index 268510e..e918e0b 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -31,10 +31,13 @@ Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**, verbindliche Produktionspfad. - **Community UI · experimentell** öffnet `fspecii/ace-step-ui`. Die CPU-leichte React/Express-Anwendung hält Bibliothek, Playlists und - Einstellungen in `/data/music/ace-step-ui`. Ein noch nicht übernommener - Upstream-Kompatibilitätsfix für die aktuelle 72-Felder-Gradio-API ist lokal - zurückportiert; normale Generierung funktioniert, Cover und Remix gelten bis - zu eigenen Ende-zu-Ende-Tests weiterhin als experimentell. + Einstellungen in `/data/music/ace-step-ui`. Sie verwendet die offizielle + `/release_task`-API mit benannten Parametern und ist damit unabhängig von der + Reihenfolge der Gradio-Felder. Normale Generierung funktioniert; Cover und + Remix gelten bis zu eigenen Ende-zu-Ende-Tests weiterhin als experimentell. + Vor dem Start zeigt sie die übertragenen Werte an. Referenzaudio beeinflusst + nur Klang und Produktion, während Quellaudio Melodie, Rhythmus und Akkorde + erhält. Die Community-Oberfläche ist im WireGuard-Netz unter `http://192.168.1.212:7861`, die originale Gradio-Oberfläche unter diff --git a/docs/SPECIALIZED_MODEL_ROADMAP.md b/docs/SPECIALIZED_MODEL_ROADMAP.md index 99e5024..9be1dc4 100644 --- a/docs/SPECIALIZED_MODEL_ROADMAP.md +++ b/docs/SPECIALIZED_MODEL_ROADMAP.md @@ -50,9 +50,11 @@ gemeldete maximale CUDA-Allokation lag bei 9,38 GiB. Es gab weder OOM noch CUDA-Fehler. Der technische Test ist damit bestanden; die subjektive Die originale ACE-Step-Gradio-Oberflaeche ist der stabile Produktionspfad. Die persistente `fspecii/ace-step-ui`-Oberflaeche bleibt bis zur Abnahme aller -Audio-zu-Audio-Modi experimentell. Der lokale Backport des offenen Upstream-PR -[#109](https://github.com/fspecii/ace-step-ui/pull/109) wurde gegen die 72 -Parameter des live veroeffentlichten `/generation_wrapper`-Schemas getestet. +Audio-zu-Audio-Modi experimentell. Am 10. September 2026 wurde der zerbrechliche +Aufruf der positionsabhaengigen `/generation_wrapper`-Schnittstelle entfernt. +Die Community-UI nutzt nun `/release_task` mit benannten Parametern; das +versionierte Worker-Derivat reicht dabei auch Referenz-/Quellaudio, Cover- +Staerke, Thinking, AI Enhance und die XL-SFT-Werte weiter. Ein zehnsekündiger FLAC-Textauftrag lief am 8. September 2026 erfolgreich durch UI, Express-Backend und Gradio-API und wurde in der persistenten Bibliothek gespeichert; dieser Test belegt Cover und Remix ausdrücklich noch nicht. diff --git a/experiments/acestep15-xl-sft/README.md b/experiments/acestep15-xl-sft/README.md index a57d8d9..a162bc9 100644 --- a/experiments/acestep15-xl-sft/README.md +++ b/experiments/acestep15-xl-sft/README.md @@ -38,15 +38,24 @@ WireGuard-Netz unter `http://192.168.1.212:7862` erreichbar. `music-ui` baut [fspecii/ace-step-ui](https://github.com/fspecii/ace-step-ui) reproduzierbar von Commit `a1fdf91829ec6f7b98844f80e323529cd155dbf2`. -Ein lokaler Backport des noch offenen Upstream-PR -[#109](https://github.com/fspecii/ace-step-ui/pull/109) gleicht das Frontend an -das aktuelle 72-Felder-Gradio-Schema an. Athena-spezifisch sind nur die -persistente Ablage, die XL-SFT-Anzeige und die gemeldeten Laufzeitlimits. +Die Community-Oberflaeche verwendet nicht mehr das positionsabhaengige +Gradio-Schema. Ihr Express-Dienst ruft die offizielle `/release_task`-API mit +benannten Feldern auf; das kleine Worker-Derivat erweitert diese Route um die +im installierten `GenerationParams` bereits vorhandenen Felder fuer Referenz-, +Quell- und Coveraudio sowie XL-SFT-Parameter. Athena-spezifisch sind ausserdem +die persistente Ablage, die XL-SFT-Anzeige und die gemeldeten Laufzeitlimits. Schlaegt bei einer spaeteren Upstream-Fassung ein Patch-Anker fehl, bricht der Image-Build ab. Die Community-UI unter `http://192.168.1.212:7861` ist bis zu vollstaendigen Ende-zu-Ende-Tests von Cover und Remix als experimentell gekennzeichnet. +Die Community-UI startet XL-SFT mit 50 Schritten, Guidance 7, Shift 1, FLAC +und aktivem Thinking ueber den 1,7B-Planer. `AI Enhance` wird als `use_format` +uebertragen. Vor jedem Auftrag zeigt sie die tatsaechlich gesendeten Parameter +und faengt offensichtliche Widersprueche ab. Referenzaudio steuert nur Klang +und Produktion; nur **Quellaudio / Cover** erhaelt Melodie, Rhythmus und +Akkorde. + ## Start Vor dem Start muessen das aktive llama.cpp-Profil und Qwen3-TTS beendet sein. diff --git a/experiments/acestep15-xl-sft/ace-step-ui/patch-source.mjs b/experiments/acestep15-xl-sft/ace-step-ui/patch-source.mjs index bf27443..fedc0db 100644 --- a/experiments/acestep15-xl-sft/ace-step-ui/patch-source.mjs +++ b/experiments/acestep15-xl-sft/ace-step-ui/patch-source.mjs @@ -27,98 +27,202 @@ patch('server/src/services/acestep.ts', (text) => { 'persistent generated audio', ); - const oldArgs = ` params.instruction || 'Fill the audio semantic mask with the style described in the text prompt.', // 17: Instruction - params.audioCoverStrength ?? 1.0, // 18: Audio Cover Strength - 0.0, // 19: Cover Noise Strength (ACE-Step v1.5 new param, default 0.0) - (params.taskType === 'audio2audio' ? 'cover' : params.taskType) || 'text2music', // 20: Task Type - params.useAdg ?? false, // 21: Use ADG - params.cfgIntervalStart ?? 0.0, // 22: CFG Interval Start - params.cfgIntervalEnd ?? 1.0, // 23: CFG Interval End - params.shift ?? 3.0, // 24: Shift - params.inferMethod || 'ode', // 25: Inference Method - params.customTimesteps || '', // 26: Custom Timesteps - params.audioFormat || 'mp3', // 27: Audio Format - params.lmTemperature ?? 0.85, // 28: LM Temperature - isThinking, // 29: Think - params.lmCfgScale ?? 2.0, // 30: LM CFG Scale - params.lmTopK ?? 0, // 31: LM Top-K - params.lmTopP ?? 0.9, // 32: LM Top-P - params.lmNegativePrompt || 'NO USER INPUT', // 33: LM Negative Prompt - useCot ? (params.useCotMetas ?? true) : false, // 34: CoT Metas - useCot ? (params.useCotCaption ?? true) : false, // 35: CaptionRewrite - useCot ? (params.useCotLanguage ?? true) : false, // 36: CoT Language - params.isFormatCaption ?? false, // 37: Is Format Caption State - params.constrainedDecodingDebug ?? false, // 38: Constrained Decoding Debug - params.allowLmBatch ?? true, // 39: ParallelThinking - params.getScores ?? false, // 40: Auto Score - params.getLrc ?? false, // 41: Auto LRC (timestamped lyrics) - params.scoreScale ?? 0.5, // 42: Quality Score Sensitivity (0.01-1.0) - params.lmBatchChunkSize ?? 8, // 43: LM Batch Chunk Size - params.trackName || null, // 44: Track Name - params.completeTrackClasses || [], // 45: Track Names - true, // 46: Enable Normalization (ACE-Step v1.5, default true) - -1.0, // 47: Normalization DB (ACE-Step v1.5, default -1.0) - 0.0, // 48: Latent Shift (ACE-Step v1.5, default 0.0) - 1.0, // 49: Latent Rescale (ACE-Step v1.5, default 1.0) - params.autogen ?? false, // 50: AutoGen`; + text = text.replace("import { handle_file } from '@gradio/client';\n", ''); + text = text.replace('getGradioClient, ', ''); - const newArgs = ` params.instruction || 'Fill the audio semantic mask based on the given conditions:', // 17: Instruction - params.audioCoverStrength ?? 1.0, // 18: LM Codes Strength - 0.0, // 19: Cover Strength - false, // 20: no_fsq - params.useAdg ?? false, // 21: Use ADG - params.cfgIntervalStart ?? 0.0, // 22: CFG Interval Start - params.cfgIntervalEnd ?? 1.0, // 23: CFG Interval End - params.shift ?? 1.0, // 24: Shift - params.inferMethod || 'ode', // 25: Inference Method - 'euler', // 26: Sampler Mode - 0.0, // 27: Velocity Norm Threshold - 0.0, // 28: Velocity EMA Factor - false, // 29: Enable DCW - 'double', // 30: DCW Mode - 0.02, // 31: DCW Scaler - 0.06, // 32: DCW High Scaler - 'haar', // 33: DCW Wavelet - params.customTimesteps || '', // 34: Custom Timesteps - params.audioFormat || 'flac', // 35: Audio Format - '320k', // 36: MP3 Bitrate - 48000, // 37: MP3 Sample Rate - params.lmTemperature ?? 0.85, // 38: LM Temperature - isThinking, // 39: Think - params.lmCfgScale ?? 2.0, // 40: LM CFG Scale - params.lmTopK ?? 0, // 41: LM Top-K - params.lmTopP ?? 0.9, // 42: LM Top-P - params.lmNegativePrompt || 'NO USER INPUT', // 43: LM Negative Prompt - useCot ? (params.useCotMetas ?? true) : false, // 44: CoT Metas - useCot ? (params.useCotCaption ?? true) : false, // 45: CaptionRewrite - useCot ? (params.useCotLanguage ?? true) : false, // 46: CoT Language - params.constrainedDecodingDebug ?? false, // 47: Constrained Decoding Debug - params.allowLmBatch ?? true, // 48: ParallelThinking - params.getScores ?? false, // 49: Auto Score - params.getLrc ?? false, // 50: Auto LRC - params.scoreScale ?? 0.5, // 51: Quality Score Sensitivity - params.lmBatchChunkSize ?? 8, // 52: LM Batch Chunk Size - params.trackName || null, // 53: Track Name - params.completeTrackClasses || [], // 54: Track Names - true, // 55: Enable Normalization - -1.0, // 56: Target Peak (dB) - 0.0, // 57: Fade In - 0.0, // 58: Fade Out - 0.0, // 59: Latent Shift - 1.0, // 60: Latent Rescale - 'balanced', // 61: Repaint Mode - 0.5, // 62: Repaint Strength - 0.0, // 63: variance - '', // 64: repaint seed - false, // 65: Edit - null, // 66: source caption - null, // 67: source lyrics - 0.0, // 68: n_min - 1.0, // 69: n_max - 1, // 70: n_avg - params.autogen ?? false, // 71: AutoGen`; + const helperStart = text.indexOf('// Gradio generation: map params'); + const helperEnd = text.indexOf('/**\n * Download a Gradio audio result file', helperStart); + if (helperStart < 0 || helperEnd < 0) throw new Error('legacy Gradio helper anchors missing'); + const helperCommentStart = text.lastIndexOf('// ---------------------------------------------------------------------------', helperStart); + const namedHelpers = `// --------------------------------------------------------------------------- +// Named REST generation through ACE-Step's official /release_task endpoint +// --------------------------------------------------------------------------- - return replaceOnce(text, oldArgs, newArgs, 'Athena Gradio argument schema'); +function resolveAudioPath(audioUrl: string): string { + if (audioUrl.startsWith('/audio/')) { + return path.join(AUDIO_DIR, audioUrl.replace('/audio/', '')); + } + if (audioUrl.startsWith('http')) { + try { + const parsed = new URL(audioUrl); + if (parsed.pathname.startsWith('/audio/')) { + return path.join(AUDIO_DIR, parsed.pathname.replace('/audio/', '')); + } + } catch { /* fall through */ } + } + return audioUrl; +} + +function resolveWorkerAudioPath(audioUrl: string | undefined): string | undefined { + if (!audioUrl) return undefined; + const localPath = resolveAudioPath(audioUrl); + if (!existsSync(localPath)) { + throw new Error(\`Uploaded audio is missing: \${localPath}\`); + } + const relativePath = path.relative(AUDIO_DIR, localPath); + if (relativePath.startsWith('..') || path.isAbsolute(relativePath)) { + throw new Error('Audio path is outside the shared Community UI storage'); + } + return path.posix.join('/data/community-audio', relativePath.split(path.sep).join('/')); +} + +function buildReleaseTaskPayload(params: GenerationParams): Record { + const caption = params.style || 'pop music'; + const prompt = params.customMode ? caption : (params.songDescription || caption); + const thinking = params.thinking ?? true; + const enhance = params.enhance ?? false; + const taskType = params.taskType === 'audio2audio' ? 'cover' : (params.taskType || 'text2music'); + + return { + prompt, + lyrics: params.instrumental ? '[Instrumental]' : (params.lyrics || ''), + instrumental: params.instrumental, + vocal_language: params.vocalLanguage || 'en', + bpm: params.bpm && params.bpm > 0 ? params.bpm : 0, + key_scale: params.keyScale || '', + time_signature: params.timeSignature || '', + audio_duration: params.duration && params.duration > 0 ? params.duration : -1, + inference_steps: params.inferenceSteps ?? 50, + guidance_scale: params.guidanceScale ?? 7.0, + shift: params.shift ?? 1.0, + infer_method: params.inferMethod || 'ode', + batch_size: Math.min(Math.max(params.batchSize ?? 1, 1), 16), + use_random_seed: params.randomSeed !== false, + seed: params.seed ?? -1, + thinking, + use_format: enhance, + lm_temperature: params.lmTemperature ?? 0.85, + lm_cfg_scale: params.lmCfgScale ?? 2.0, + lm_top_k: params.lmTopK ?? 0, + lm_top_p: params.lmTopP ?? 0.9, + lm_negative_prompt: params.lmNegativePrompt || 'NO USER INPUT', + use_cot_metas: thinking ? (params.useCotMetas ?? true) : false, + use_cot_caption: thinking ? (params.useCotCaption ?? true) : false, + use_cot_language: thinking ? (params.useCotLanguage ?? true) : false, + allow_lm_batch: params.allowLmBatch ?? true, + constrained_decoding_debug: params.constrainedDecodingDebug ?? false, + lm_batch_chunk_size: params.lmBatchChunkSize ?? 8, + task_type: taskType, + instruction: params.instruction || 'Fill the audio semantic mask based on the given conditions:', + reference_audio_path: resolveWorkerAudioPath(params.referenceAudioUrl), + src_audio_path: resolveWorkerAudioPath(params.sourceAudioUrl), + audio_codes: params.audioCodes || '', + repainting_start: params.repaintingStart ?? 0.0, + repainting_end: params.repaintingEnd ?? -1, + audio_cover_strength: params.audioCoverStrength ?? 1.0, + use_adg: params.useAdg ?? false, + cfg_interval_start: params.cfgIntervalStart ?? 0.0, + cfg_interval_end: params.cfgIntervalEnd ?? 1.0, + audio_format: params.audioFormat || 'flac', + mp3_bitrate: '320k', + mp3_sample_rate: 48000, + }; +} + +`; + text = text.slice(0, helperCommentStart) + namedHelpers + text.slice(helperEnd); + + const processStart = text.indexOf('// processGeneration — Gradio primary'); + const processEnd = text.indexOf('function isAudioFile', processStart); + if (processStart < 0 || processEnd < 0) throw new Error('legacy generation anchors missing'); + const processCommentStart = text.lastIndexOf('// ---------------------------------------------------------------------------', processStart); + const namedProcess = `// --------------------------------------------------------------------------- +// processGeneration — official named REST API only +// --------------------------------------------------------------------------- + +async function processGeneration( + jobId: string, + params: GenerationParams, + job: ActiveJob, +): Promise { + job.status = 'running'; + job.stage = 'Preparing named ACE-Step request...'; + + if ((params.taskType === 'cover' || params.taskType === 'audio2audio') && !params.sourceAudioUrl && !params.audioCodes) { + job.status = 'failed'; + job.error = \`task_type='\${params.taskType}' requires source audio or audio codes\`; + return; + } + + try { + if (params.ditModel) { + job.stage = \`Loading model \${params.ditModel}...\`; + await switchModelIfNeeded(params.ditModel); + } + + const payload = buildReleaseTaskPayload(params); + console.log(\`Job \${jobId}: POST /release_task with named parameters\`, payload); + job.stage = 'Generating music via named ACE-Step API...'; + + const releaseResponse = await fetch(\`\${ACESTEP_API}/release_task\`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(payload), + }); + const releaseText = await releaseResponse.text(); + if (!releaseResponse.ok) { + throw new Error(\`ACE-Step /release_task failed (\${releaseResponse.status}): \${releaseText}\`); + } + const release = JSON.parse(releaseText) as any; + if (release.code !== 200 || !release.data?.task_id) { + throw new Error(release.error || 'ACE-Step returned no task_id'); + } + + const taskId = String(release.data.task_id); + job.taskId = taskId; + const queryResponse = await fetch(\`\${ACESTEP_API}/query_result\`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ task_id_list: [taskId] }), + }); + const queryText = await queryResponse.text(); + if (!queryResponse.ok) { + throw new Error(\`ACE-Step /query_result failed (\${queryResponse.status}): \${queryText}\`); + } + const query = JSON.parse(queryText) as any; + const taskResult = query.data?.[0]; + const audioItems = taskResult?.result ? JSON.parse(taskResult.result) : []; + if (!Array.isArray(audioItems) || audioItems.length === 0) { + throw new Error('ACE-Step completed without downloadable audio results'); + } + + const audioUrls: string[] = []; + let actualDuration = 0; + for (const item of audioItems) { + if (!item?.url) continue; + const remoteUrl = new URL(item.url, ACESTEP_API).toString(); + const remoteName = String(item.file || item.url); + const ext = path.extname(remoteName) || \`.\${params.audioFormat || 'flac'}\`; + const filename = \`\${jobId}_\${audioUrls.length}\${ext}\`; + const destPath = path.join(AUDIO_DIR, filename); + await downloadGradioAudioFile({ url: remoteUrl, orig_name: remoteName }, destPath); + if (audioUrls.length === 0) actualDuration = getAudioDuration(destPath); + audioUrls.push(\`/audio/\${filename}\`); + } + if (audioUrls.length === 0) throw new Error('ACE-Step returned no supported audio files'); + + const first = audioItems[0] || {}; + job.status = 'succeeded'; + job.result = { + audioUrls, + duration: actualDuration || Number(first.duration) || params.duration || 0, + bpm: Number(first.bpm) || params.bpm, + keyScale: first.keyscale || params.keyScale, + timeSignature: first.timesignature || params.timeSignature, + status: 'succeeded', + }; + job.rawResponse = { release, query, transmittedParameters: payload }; + console.log(\`Job \${jobId}: Completed via named REST API with \${audioUrls.length} audio files\`); + } catch (error) { + job.status = 'failed'; + job.error = error instanceof Error ? error.message : String(error); + console.error(\`Job \${jobId}: Named REST generation failed\`, error); + } +} + +`; + text = text.slice(0, processCommentStart) + namedProcess + text.slice(processEnd); + return text; }); patch('server/src/services/storage/local.ts', (text) => { @@ -159,6 +263,17 @@ patch('server/src/routes/referenceTrack.ts', (text) => { }); patch('server/src/routes/generate.ts', (text) => { + text = replaceOnce( + text, + " thinking?: boolean;\n audioFormat?: 'mp3' | 'flac';", + " thinking?: boolean;\n enhance?: boolean;\n audioFormat?: 'mp3' | 'flac';", + 'server enhance type', + ); + const enhanceAnchor = ' thinking,\n audioFormat,'; + if (text.split(enhanceAnchor).length - 1 !== 2) { + throw new Error('expected enhance anchor in destructuring and forwarding'); + } + text = text.replaceAll(enhanceAnchor, ' thinking,\n enhance,\n audioFormat,'); text = replaceOnce( text, " const ALL_DIT_MODELS = [\n 'acestep-v15-turbo',", @@ -185,3 +300,108 @@ patch('server/src/routes/generate.ts', (text) => { `; return text.slice(0, start) + limits + text.slice(end); }); + +patch('services/api.ts', (text) => replaceOnce( + text, + " thinking?: boolean;\n audioFormat?: 'mp3' | 'flac';", + " thinking?: boolean;\n enhance?: boolean;\n audioFormat?: 'mp3' | 'flac';", + 'client enhance type', +)); + +patch('App.tsx', (text) => { + text = replaceOnce( + text, + ' thinking: params.thinking,\n audioFormat: params.audioFormat,', + ' thinking: params.thinking,\n enhance: params.enhance,\n audioFormat: params.audioFormat,', + 'client enhance forwarding', + ); + return replaceOnce( + text, + ' title: params.title,\n instrumental: params.instrumental,', + ' title: params.title,\n ditModel: params.ditModel,\n instrumental: params.instrumental,', + 'client model forwarding', + ); +}); + +patch('components/CreatePanel.tsx', (text) => { + text = replaceOnce(text, 'useState(9.0);', 'useState(7.0);', 'guidance default'); + text = replaceOnce( + text, + 'useState(false); // Default false for GPU compatibility', + 'useState(true); // Athena default: use the 1.7B planner for coherent structure', + 'thinking default', + ); + text = replaceOnce(text, "useState<'mp3' | 'flac'>('mp3');", "useState<'mp3' | 'flac'>('flac');", 'lossless default'); + text = replaceOnce(text, 'useState(12);', 'useState(50);', 'XL-SFT steps default'); + text = replaceOnce(text, "localStorage.getItem('ace-lmModel') || 'acestep-5Hz-lm-0.6B'", "localStorage.getItem('ace-lmModel') || 'acestep-5Hz-lm-1.7B'", 'planner model default'); + text = replaceOnce(text, 'useState(3.0);', 'useState(1.0);', 'shift default'); + + text = replaceOnce( + text, + ' // Bulk generation: loop bulkCount times\n for (let i = 0; i < bulkCount; i++) {', + ` const requestedText = customMode ? styleWithGender : songDescription; + const asksForVocals = /\\b(vocals?|singer|singing|male voice|female voice|gesang|stimme|sänger(?:in)?|singt)\\b/i.test( + \`\${requestedText || ''}\\n\${lyrics}\`, + ); + if (!instrumental && asksForVocals && !lyrics.trim()) { + window.alert('Widerspruch: Der Auftrag verlangt Gesang, aber das Liedtextfeld ist leer. Bitte Text eintragen oder „Instrumental“ wählen.'); + return; + } + if ((taskType === 'cover' || taskType === 'audio2audio') && !sourceAudioUrl.trim() && !audioCodes.trim()) { + window.alert('Für einen Cover-Auftrag fehlt das Quellaudio. Bitte unter „Quellaudio / Cover“ eine Datei auswählen.'); + return; + } + + const taskLabel = taskType === 'cover' || taskType === 'audio2audio' ? 'Cover / Audio-zu-Audio' : taskType; + const summary = [ + 'Folgende Parameter werden tatsächlich an Athena übertragen:', + '', + \`Aufgabe: \${taskLabel}\`, + \`Modell: \${selectedModel}\`, + \`Dauer: \${duration > 0 ? \`\${duration} Sekunden\` : 'automatisch'}\`, + \`Tempo: \${bpm > 0 ? \`\${bpm} BPM\` : 'automatisch'}\`, + \`Tonart: \${keyScale || 'automatisch'}\`, + \`Taktart: \${timeSignature || 'automatisch'}\`, + \`Thinking/Planung: \${thinking ? 'AN' : 'AUS'}\`, + \`AI Enhance: \${enhance ? 'AN' : 'AUS'}\`, + \`XL-SFT: \${inferenceSteps} Schritte, Guidance \${guidanceScale}, Shift \${shift}\`, + \`Gesang: \${instrumental ? 'nein (Instrumental)' : 'ja'}\`, + \`Referenzaudio (nur Klang/Produktion): \${referenceAudioUrl ? 'vorhanden' : 'keines'}\`, + \`Quellaudio (Melodie/Rhythmus/Akkorde): \${sourceAudioUrl ? 'vorhanden' : 'keines'}\`, + \`Ausgabe: \${audioFormat.toUpperCase()}, \${batchSize} Variation(en), \${bulkCount} Auftrag/Aufträge\`, + '', + 'Auftrag jetzt starten?', + ].join('\\n'); + if (!window.confirm(summary)) return; + + // Bulk generation: loop bulkCount times + for (let i = 0; i < bulkCount; i++) {`, + 'validation and transmitted parameter summary', + ); + + text = replaceOnce( + text, + " {t('reference')}\n ", + " Referenzaudio\n ", + 'reference tab label', + ); + text = replaceOnce( + text, + " {t('cover')}\n ", + " Quellaudio / Cover\n ", + 'source tab label', + ); + text = replaceOnce( + text, + ' {/* Audio Content */}\n
', + ` {/* Audio Content */} +
+

+ {audioTab === 'reference' + ? 'Referenzaudio beeinflusst nur Klang, Instrumentierung und Produktion – nicht die Melodie.' + : 'Quellaudio / Cover erhält Melodie, Rhythmus und Akkorde des hochgeladenen Titels.'} +

`, + 'audio semantics explanation', + ); + return text; +}); diff --git a/experiments/acestep15-xl-sft/compose.yaml b/experiments/acestep15-xl-sft/compose.yaml index 5b26c70..19967be 100644 --- a/experiments/acestep15-xl-sft/compose.yaml +++ b/experiments/acestep15-xl-sft/compose.yaml @@ -1,6 +1,8 @@ services: music-worker: - image: ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567 + build: + context: ./worker + image: mike-ai/ace-step-1.5:named-api-v1 container_name: mike-ai-music-acestep-test labels: com.mike-ai.music-worker: "acestep" @@ -34,6 +36,8 @@ services: - ${ACESTEP_HF_CACHE_DIR:-/data/models/acestep/hf-cache}:/root/.cache/huggingface - ${ACESTEP_OUTPUT_DIR:-/data/music/acestep}:/app/gradio_outputs - ${ACESTEP_OUTPUT_DIR:-/data/music/acestep}:/app/output + # Uploaded Community-UI audio is shared read-only with the named REST API. + - ${ACESTEP_UI_DATA_DIR:-/data/music/ace-step-ui}/audio:/data/community-audio:ro # Version-pinned UI defaults for XL-SFT quality and lossless output. - ./overrides/model_config.py:/app/acestep/ui/gradio/events/generation/model_config.py:ro - ./overrides/generation_advanced_output_controls.py:/app/acestep/ui/gradio/interfaces/generation_advanced_output_controls.py:ro diff --git a/experiments/acestep15-xl-sft/worker/Dockerfile b/experiments/acestep15-xl-sft/worker/Dockerfile new file mode 100644 index 0000000..31ded2a --- /dev/null +++ b/experiments/acestep15-xl-sft/worker/Dockerfile @@ -0,0 +1,6 @@ +FROM ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567 + +COPY patch-api-routes.py /tmp/patch-api-routes.py +RUN /usr/bin/python3 /tmp/patch-api-routes.py \ + /app/acestep/ui/gradio/api/api_routes.py \ + && rm /tmp/patch-api-routes.py diff --git a/experiments/acestep15-xl-sft/worker/patch-api-routes.py b/experiments/acestep15-xl-sft/worker/patch-api-routes.py new file mode 100644 index 0000000..2d43135 --- /dev/null +++ b/experiments/acestep15-xl-sft/worker/patch-api-routes.py @@ -0,0 +1,137 @@ +"""Extend ACE-Step's official /release_task route with named generation inputs. + +The base image already provides the route. This build-time patch only exposes +the parameters supported by its installed GenerationParams/GenerationConfig +dataclasses, so the separate Community UI never has to depend on Gradio's +positional component order. +""" + +from pathlib import Path +import sys + + +target = Path(sys.argv[1]) +source = target.read_text(encoding="utf-8") + + +def replace_once(old: str, new: str, label: str) -> None: + global source + count = source.count(old) + if count != 1: + raise RuntimeError(f"{label}: expected one anchor, found {count}") + source = source.replace(old, new, 1) + + +old_params = ''' # Build generation params with alias support + params = GenerationParams( + task_type=get_param("task_type", default="text2music"), + caption=caption, + lyrics=lyrics, + bpm=sample_bpm or get_param("bpm"), + keyscale=sample_keyscale or get_param("key_scale", "keyscale", "key", default=""), + timesignature=sample_timesignature or get_param("time_signature", "timesignature", default=""), + duration=sample_duration or get_param("audio_duration", "duration", default=-1), + vocal_language=sample_language, + inference_steps=get_param("inference_steps", default=8), + guidance_scale=float(get_param("guidance_scale", default=7.0) or 7.0), + seed=int(get_param("seed", default=-1) or -1), + thinking=to_bool(get_param("thinking"), False), + lm_temperature=lm_temperature, + lm_cfg_scale=float(get_param("lm_cfg_scale", default=2.0) or 2.0), + lm_negative_prompt=get_param("lm_negative_prompt", default="NO USER INPUT") or "NO USER INPUT", + repaint_latent_crossfade_frames=int( + get_param("repaint_latent_crossfade_frames", default=10) or 10, + ), + repaint_wav_crossfade_sec=float( + get_param("repaint_wav_crossfade_sec", default=0.0) or 0.0, + ), + repaint_mode=get_param("repaint_mode", default="balanced") or "balanced", + repaint_strength=float( + get_param("repaint_strength", default=0.5) or 0.5, + ), + ) +''' + +new_params = ''' # Build generation params with alias support. Keep this + # mapping explicit: every public API field below is named and independent + # from the order of components in the Gradio interface. + raw_bpm = sample_bpm or get_param("bpm") + params = GenerationParams( + task_type=get_param("task_type", default="text2music") or "text2music", + instruction=get_param("instruction", default="Fill the audio semantic mask based on the given conditions:") or "Fill the audio semantic mask based on the given conditions:", + reference_audio=get_param("reference_audio_path", "reference_audio"), + src_audio=get_param("src_audio_path", "src_audio", "source_audio"), + audio_codes=get_param("audio_codes", default="") or "", + caption=caption, + lyrics=lyrics, + instrumental=to_bool(get_param("instrumental"), False), + bpm=int(float(raw_bpm)) if raw_bpm not in (None, "", 0, "0") else None, + keyscale=sample_keyscale or get_param("key_scale", "keyscale", "key", default=""), + timesignature=sample_timesignature or get_param("time_signature", "timesignature", default=""), + duration=float(sample_duration or get_param("audio_duration", "duration", default=-1) or -1), + vocal_language=sample_language, + inference_steps=int(get_param("inference_steps", default=50) or 50), + guidance_scale=float(get_param("guidance_scale", default=7.0) or 7.0), + seed=int(get_param("seed", default=-1) or -1), + use_adg=to_bool(get_param("use_adg"), False), + cfg_interval_start=float(get_param("cfg_interval_start", default=0.0) or 0.0), + cfg_interval_end=float(get_param("cfg_interval_end", default=1.0) or 1.0), + shift=float(get_param("shift", default=1.0) or 1.0), + infer_method=get_param("infer_method", default="ode") or "ode", + sampler_mode=get_param("sampler_mode", default="euler") or "euler", + repainting_start=float(get_param("repainting_start", default=0.0) or 0.0), + repainting_end=float(get_param("repainting_end", default=-1.0) or -1.0), + chunk_mask_mode=get_param("chunk_mask_mode", default="auto") or "auto", + audio_cover_strength=float(get_param("audio_cover_strength", default=1.0) or 1.0), + cover_noise_strength=float(get_param("cover_noise_strength", default=0.0) or 0.0), + thinking=to_bool(get_param("thinking"), True), + lm_temperature=lm_temperature, + lm_cfg_scale=float(get_param("lm_cfg_scale", default=2.0) or 2.0), + lm_top_k=int(get_param("lm_top_k", default=0) or 0), + lm_top_p=float(get_param("lm_top_p", default=0.9) or 0.9), + lm_negative_prompt=get_param("lm_negative_prompt", default="NO USER INPUT") or "NO USER INPUT", + use_cot_metas=to_bool(get_param("use_cot_metas"), True), + use_cot_caption=to_bool(get_param("use_cot_caption"), True), + use_cot_lyrics=to_bool(get_param("use_cot_lyrics"), False), + use_cot_language=to_bool(get_param("use_cot_language"), True), + use_constrained_decoding=to_bool(get_param("use_constrained_decoding"), True), + enable_normalization=to_bool(get_param("enable_normalization"), True), + normalization_db=float(get_param("normalization_db", default=-1.0) or -1.0), + fade_in_duration=float(get_param("fade_in_duration", default=0.0) or 0.0), + fade_out_duration=float(get_param("fade_out_duration", default=0.0) or 0.0), + latent_shift=float(get_param("latent_shift", default=0.0) or 0.0), + latent_rescale=float(get_param("latent_rescale", default=1.0) or 1.0), + repaint_latent_crossfade_frames=int(get_param("repaint_latent_crossfade_frames", default=10) or 10), + repaint_wav_crossfade_sec=float(get_param("repaint_wav_crossfade_sec", default=0.0) or 0.0), + repaint_mode=get_param("repaint_mode", default="balanced") or "balanced", + repaint_strength=float(get_param("repaint_strength", default=0.5) or 0.5), + ) +''' + +replace_once(old_params, new_params, "GenerationParams mapping") + +old_config = ''' config = GenerationConfig( + batch_size=get_param("batch_size", default=2), + use_random_seed=use_random_seed, + seeds=resolved_seeds, + audio_format=get_param("audio_format", default="flac"), + mp3_bitrate=get_param("mp3_bitrate", default="128k"), + mp3_sample_rate=get_param("mp3_sample_rate", default=48000), + ) +''' + +new_config = ''' config = GenerationConfig( + batch_size=int(get_param("batch_size", default=1) or 1), + allow_lm_batch=to_bool(get_param("allow_lm_batch"), True), + use_random_seed=to_bool(use_random_seed, True), + seeds=resolved_seeds, + lm_batch_chunk_size=int(get_param("lm_batch_chunk_size", default=8) or 8), + constrained_decoding_debug=to_bool(get_param("constrained_decoding_debug"), False), + audio_format=get_param("audio_format", default="flac") or "flac", + mp3_bitrate=get_param("mp3_bitrate", default="320k") or "320k", + mp3_sample_rate=int(get_param("mp3_sample_rate", default=48000) or 48000), + ) +''' + +replace_once(old_config, new_config, "GenerationConfig mapping") +target.write_text(source, encoding="utf-8")