48 lines
1.3 KiB
Python
48 lines
1.3 KiB
Python
#!/usr/bin/env python3
|
|
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import os
|
|
import time
|
|
|
|
import torch
|
|
|
|
import models.svc.vevo2.infer_vevo2_fm as vevo
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("--source", required=True)
|
|
parser.add_argument("--reference", required=True)
|
|
parser.add_argument("--output", required=True)
|
|
parser.add_argument("--no-pitch-shift", action="store_true")
|
|
args = parser.parse_args()
|
|
|
|
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
|
|
started = time.monotonic()
|
|
vevo.inference_pipeline = vevo.load_inference_pipeline()
|
|
loaded = time.monotonic()
|
|
vevo.vevo2_fm(
|
|
args.source,
|
|
args.reference,
|
|
args.output,
|
|
shifted_src=not args.no_pitch_shift,
|
|
)
|
|
finished = time.monotonic()
|
|
print(
|
|
{
|
|
"model_load_seconds": round(loaded - started, 3),
|
|
"conversion_seconds": round(finished - loaded, 3),
|
|
"total_seconds": round(finished - started, 3),
|
|
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
|
|
"output": args.output,
|
|
},
|
|
flush=True,
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|