#!/usr/bin/env python3 """Minimal reproducible Vevo2 style-preserving voice-conversion smoke test.""" from __future__ import annotations import argparse import os import time import torch import models.svc.vevo2.infer_vevo2_fm as vevo def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--source", required=True) parser.add_argument("--reference", required=True) parser.add_argument("--output", required=True) parser.add_argument("--no-pitch-shift", action="store_true") args = parser.parse_args() os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True) started = time.monotonic() vevo.inference_pipeline = vevo.load_inference_pipeline() loaded = time.monotonic() vevo.vevo2_fm( args.source, args.reference, args.output, shifted_src=not args.no_pitch_shift, ) finished = time.monotonic() print( { "model_load_seconds": round(loaded - started, 3), "conversion_seconds": round(finished - loaded, 3), "total_seconds": round(finished - started, 3), "peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1), "output": args.output, }, flush=True, ) if __name__ == "__main__": main()