feat: add full 4-stem demucs mode with MIDI instrument replacement

- Support mode='full' in handle_demucs_separating config
- Full mode: 4-stem separation → basic-pitch MIDI → fluidsynth render → mix
- Each instrument gets correct GM instrument (drums=channel10, bass=34, other=1)
- 2-stem mode preserved as default for backward compatibility
- Uses run_full.py standalone script on GPU server via SSH
This commit is contained in:
yumoqing 2026-07-06 11:30:43 +08:00
parent 0bc74fdd1a
commit ffdba55fd7

View File

@ -156,9 +156,15 @@ async def handle_video_preparing(tenant_id, task_id, step_name, input_data, conf
# ─── Demucs Separation ───────────────────────────────────────────────
async def handle_demucs_separating(tenant_id, task_id, step_name, input_data, config):
"""Run Demucs on GPU server to separate vocals and accompaniment."""
"""Run Demucs on GPU server to separate vocals and accompaniment.
Supports two modes via config.mode:
- "2stem" (default): classic vocals + no_vocals
- "full": 4-stem MIDI replace instruments mix accompaniment
"""
work_dir = _task_dir(task_id)
gpu_dir = _gpu_task_dir(task_id)
mode = config.get("mode", "2stem")
# Find audio path from deps
audio_path = None
@ -178,7 +184,18 @@ async def handle_demucs_separating(tenant_id, task_id, step_name, input_data, co
remote_audio = f"{gpu_dir}/audio.mp3"
await _copy_to_gpu(audio_path, remote_audio)
# Run Demucs on GPU
if mode == "full":
# ── 4-stem + MIDI replace pipeline ──────────────────────────
result = await _demucs_full_pipeline(task_id, remote_audio, gpu_dir, work_dir)
else:
# ── Classic 2-stem pipeline ─────────────────────────────────
result = await _demucs_2stem_pipeline(remote_audio, gpu_dir, work_dir)
return result
async def _demucs_2stem_pipeline(remote_audio, gpu_dir, work_dir):
"""Classic 2-stem vocal separation."""
demucs_cmd = (
f"cd {gpu_dir} && "
f"source {GPU_DEMUCS_VENV}/bin/activate && "
@ -189,7 +206,6 @@ async def handle_demucs_separating(tenant_id, task_id, step_name, input_data, co
if rc != 0:
raise ValueError(f"Demucs 分离失败: {stderr}")
# Copy results back
vocals_local = os.path.join(work_dir, "vocals.wav")
no_vocals_local = os.path.join(work_dir, "no_vocals.wav")
base = os.path.splitext(os.path.basename(remote_audio))[0]
@ -199,6 +215,60 @@ async def handle_demucs_separating(tenant_id, task_id, step_name, input_data, co
return {
"vocals_path": vocals_local,
"no_vocals_path": no_vocals_local,
"mode": "2stem",
}
async def _demucs_full_pipeline(task_id, remote_audio, gpu_dir, work_dir):
"""4-stem separation → MIDI instrument replacement → accompaniment mix.
Runs the standalone run_full.py on GPU server via SSH.
"""
import json as _json
output_dir = f"{gpu_dir}/full_output"
cmd = (
f"cd /data/ymq/demucs-service && "
f"DEMUCS_GPU_ID=5 "
f"{GPU_DEMUCS_VENV}/bin/python run_full.py "
f"'{remote_audio}' '{output_dir}'"
)
logger.info(f"[demucs-full] Starting full pipeline for task {task_id}")
stdout, stderr, rc = await _run_gpu(cmd, timeout=900)
if rc != 0:
raise ValueError(f"Demucs full separation failed: {stderr[-500:]}")
# Parse JSON result from stdout
try:
gpu_result = _json.loads(stdout)
except _json.JSONDecodeError:
# Try to extract JSON from mixed output
lines = stdout.strip().split("\n")
for line in reversed(lines):
try:
gpu_result = _json.loads(line)
break
except _json.JSONDecodeError:
continue
else:
raise ValueError(f"Failed to parse demucs output: {stdout[-500:]}")
# Copy results back to local
vocals_local = os.path.join(work_dir, "vocals.wav")
accompaniment_local = os.path.join(work_dir, "accompaniment.wav")
await _copy_from_gpu(gpu_result["vocals_path"], vocals_local)
await _copy_from_gpu(gpu_result["accompaniment_path"], accompaniment_local)
return {
"vocals_path": vocals_local,
"no_vocals_path": accompaniment_local, # backward compat
"accompaniment_path": accompaniment_local, # new name
"mode": "full",
"duration": gpu_result.get("duration", 0),
}