Files

48 lines
1.3 KiB
Python

#!/usr/bin/env python3
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
from __future__ import annotations
import argparse
import os
import time
import torch
import models.svc.vevo2.infer_vevo2_fm as vevo
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--source", required=True)
parser.add_argument("--reference", required=True)
parser.add_argument("--output", required=True)
parser.add_argument("--no-pitch-shift", action="store_true")
args = parser.parse_args()
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
started = time.monotonic()
vevo.inference_pipeline = vevo.load_inference_pipeline()
loaded = time.monotonic()
vevo.vevo2_fm(
args.source,
args.reference,
args.output,
shifted_src=not args.no_pitch_shift,
)
finished = time.monotonic()
print(
{
"model_load_seconds": round(loaded - started, 3),
"conversion_seconds": round(finished - loaded, 3),
"total_seconds": round(finished - started, 3),
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
"output": args.output,
},
flush=True,
)
if __name__ == "__main__":
main()