1321 lines
43 KiB
Python
Executable File
1321 lines
43 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Unified local CosyVoice production renderer for Claude and Codex plugins."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import math
|
|
import os
|
|
import platform
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from difflib import SequenceMatcher
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
PLUGIN_ROOT = Path(__file__).resolve().parent.parent
|
|
PROFILE_DIR = PLUGIN_ROOT / "profiles"
|
|
PROMPT_PREFIX = "You are a helpful assistant.<|endofprompt|>"
|
|
MODEL_RELATIVE = Path("models") / "CosyVoice3-0.5B-Candle"
|
|
ID_PATTERN = re.compile(r"^[a-zA-Z0-9][a-zA-Z0-9._-]*$")
|
|
|
|
|
|
class VoiceError(RuntimeError):
|
|
"""A production failure with an actionable message."""
|
|
|
|
|
|
def platform_default_runtime() -> Path:
|
|
configured = os.environ.get("LOCAL_VOICE_RUNTIME")
|
|
if configured:
|
|
return Path(configured).expanduser()
|
|
if platform.system() == "Windows":
|
|
base = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData" / "Local"))
|
|
return base / "MPM Local Voice" / "runtime"
|
|
if platform.system() == "Darwin":
|
|
return (
|
|
Path.home()
|
|
/ "Library"
|
|
/ "Application Support"
|
|
/ "MPM Local Voice"
|
|
/ "runtime"
|
|
)
|
|
return Path.home() / ".local" / "share" / "mpm-local-voice" / "runtime"
|
|
|
|
|
|
def json_read(path: Path) -> Any:
|
|
try:
|
|
return json.loads(path.read_text(encoding="utf-8"))
|
|
except FileNotFoundError as exc:
|
|
raise VoiceError(f"Missing file: {path}") from exc
|
|
except json.JSONDecodeError as exc:
|
|
raise VoiceError(f"Invalid JSON in {path}: {exc}") from exc
|
|
|
|
|
|
def json_write(path: Path, value: Any) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def normalized_tokens(text: str) -> list[str]:
|
|
return re.findall(r"[a-z0-9]+", text.casefold())
|
|
|
|
|
|
def word_count(text: str) -> int:
|
|
return len(normalized_tokens(text))
|
|
|
|
|
|
def clean_markdown(text: str) -> str:
|
|
text = re.sub(r"^\s{0,3}#{1,6}\s+.*$", "", text, flags=re.MULTILINE)
|
|
text = re.sub(r"^\s*[-*+]\s+", "", text, flags=re.MULTILINE)
|
|
text = re.sub(r"^\s*\d+[.)]\s+", "", text, flags=re.MULTILINE)
|
|
text = re.sub(r"^\s*>\s?", "", text, flags=re.MULTILINE)
|
|
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", "", text)
|
|
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
|
|
text = re.sub(r"[*_~`]", "", text)
|
|
return text
|
|
|
|
|
|
def paragraphs_from_markdown(path: Path) -> list[str]:
|
|
cleaned = clean_markdown(path.read_text(encoding="utf-8"))
|
|
paragraphs = [
|
|
re.sub(r"\s+", " ", block).strip()
|
|
for block in re.split(r"\n\s*\n", cleaned)
|
|
]
|
|
return [block for block in paragraphs if block]
|
|
|
|
|
|
def last_sentence(text: str, maximum_words: int = 10) -> str:
|
|
parts = re.split(r"(?<=[.!?])\s+", text.strip())
|
|
candidate = parts[-1] if parts else text.strip()
|
|
words = candidate.split()
|
|
return " ".join(words[-maximum_words:])
|
|
|
|
|
|
def first_sentence(text: str, maximum_words: int = 10) -> str:
|
|
parts = re.split(r"(?<=[.!?])\s+", text.strip())
|
|
selected: list[str] = []
|
|
for part in parts or [text.strip()]:
|
|
selected.extend(part.split())
|
|
if len(selected) >= 6:
|
|
break
|
|
return " ".join(selected[:maximum_words])
|
|
|
|
|
|
def slug(value: str) -> str:
|
|
value = re.sub(r"[^a-zA-Z0-9]+", "-", value).strip("-").casefold()
|
|
return value or "segment"
|
|
|
|
|
|
def profile_path(voice: str) -> Path:
|
|
return PROFILE_DIR / f"{voice}.json"
|
|
|
|
|
|
def load_profile(voice: str) -> dict[str, Any]:
|
|
profile = json_read(profile_path(voice))
|
|
if profile.get("id") != voice:
|
|
raise VoiceError(f"Profile ID mismatch in {profile_path(voice)}")
|
|
return profile
|
|
|
|
|
|
def available_profiles() -> list[dict[str, Any]]:
|
|
profiles = []
|
|
for path in sorted(PROFILE_DIR.glob("*.json")):
|
|
profile = json_read(path)
|
|
profiles.append(
|
|
{
|
|
"id": profile["id"],
|
|
"display_name": profile["display_name"],
|
|
"adapter": profile["adapter"],
|
|
"roles": sorted(profile["roles"]),
|
|
}
|
|
)
|
|
return profiles
|
|
|
|
|
|
def validate_plan_data(plan: dict[str, Any]) -> dict[str, Any]:
|
|
if plan.get("schema_version") != "1.0":
|
|
raise VoiceError("Render plan schema_version must be '1.0'")
|
|
job_id = str(plan.get("job_id", ""))
|
|
if not ID_PATTERN.fullmatch(job_id):
|
|
raise VoiceError("job_id must be filesystem-safe")
|
|
voice = str(plan.get("voice", ""))
|
|
profile = load_profile(voice)
|
|
output = Path(str(plan.get("output", ""))).expanduser()
|
|
if not output.is_absolute():
|
|
raise VoiceError("Render plan output must be an absolute path")
|
|
if output.suffix.casefold() not in {".wav", ".mp3"}:
|
|
raise VoiceError("Render plan output must end in .wav or .mp3")
|
|
segments = plan.get("segments")
|
|
if not isinstance(segments, list) or not segments:
|
|
raise VoiceError("Render plan must contain at least one segment")
|
|
seen: set[str] = set()
|
|
roles = set(profile["roles"])
|
|
deliveries = set(profile.get("deliveries", {}))
|
|
for index, segment in enumerate(segments, start=1):
|
|
if not isinstance(segment, dict):
|
|
raise VoiceError(f"Segment {index} must be an object")
|
|
segment_id = str(segment.get("id", ""))
|
|
if not ID_PATTERN.fullmatch(segment_id):
|
|
raise VoiceError(f"Segment {index} has invalid id {segment_id!r}")
|
|
if segment_id in seen:
|
|
raise VoiceError(f"Duplicate segment id: {segment_id}")
|
|
seen.add(segment_id)
|
|
text = str(segment.get("text", "")).strip()
|
|
if not text:
|
|
raise VoiceError(f"{segment_id}: canonical text is required")
|
|
role = segment.get("role")
|
|
if role is not None and role not in roles:
|
|
raise VoiceError(f"{segment_id}: unknown role {role!r}")
|
|
if profile["adapter"] == "dialogue" and role is None:
|
|
raise VoiceError(f"{segment_id}: dialogue segments require a role")
|
|
delivery = segment.get("delivery")
|
|
if delivery is not None and delivery not in deliveries:
|
|
raise VoiceError(f"{segment_id}: unknown delivery {delivery!r}")
|
|
fixed = segment.get("fixed_asset")
|
|
pool = segment.get("asset_pool")
|
|
if fixed and pool:
|
|
raise VoiceError(f"{segment_id}: choose fixed_asset or asset_pool")
|
|
if pool is not None and (
|
|
not isinstance(pool, list) or not all(isinstance(x, str) for x in pool)
|
|
):
|
|
raise VoiceError(f"{segment_id}: asset_pool must be a list of paths")
|
|
if segment.get("target_wpm") and segment.get("tempo_multiplier"):
|
|
raise VoiceError(
|
|
f"{segment_id}: target_wpm and tempo_multiplier are mutually exclusive"
|
|
)
|
|
return profile
|
|
|
|
|
|
def build_plan(args: argparse.Namespace) -> dict[str, Any]:
|
|
profile = load_profile(args.voice)
|
|
max_words = int(profile["generation"]["maximum_words"])
|
|
short_max = int(profile["generation"]["short_native_max_words"])
|
|
raw_blocks = paragraphs_from_markdown(args.script)
|
|
if not raw_blocks:
|
|
raise VoiceError("Script contains no speakable text")
|
|
|
|
segments: list[dict[str, Any]] = []
|
|
if profile["adapter"] == "dialogue":
|
|
role_pattern = re.compile(
|
|
r"^(?:\*\*)?(PRODUCER(?:\s+GUY)?|EXEC(?:\s+GUY)?|"
|
|
r"WRITER(?:\s+GUY)?)(?:\*\*)?\s*:\s*(.+)$",
|
|
re.IGNORECASE,
|
|
)
|
|
for block in raw_blocks:
|
|
match = role_pattern.match(block)
|
|
if not match:
|
|
raise VoiceError(
|
|
"Ryan scripts require Producer/Writer labels on every paragraph"
|
|
)
|
|
label, spoken = match.groups()
|
|
role = (
|
|
"writer"
|
|
if label.casefold().startswith("writer")
|
|
else "producer"
|
|
)
|
|
segments.append(
|
|
{
|
|
"id": f"{len(segments) + 1:02d}-{role}",
|
|
"role": role,
|
|
"text": spoken.strip(),
|
|
}
|
|
)
|
|
else:
|
|
packed: list[str] = []
|
|
for index, block in enumerate(raw_blocks):
|
|
is_short_frame = (
|
|
word_count(block) <= short_max
|
|
and index in {0, len(raw_blocks) - 1}
|
|
)
|
|
if is_short_frame:
|
|
if packed:
|
|
segments.append(
|
|
{
|
|
"id": f"{len(segments) + 1:02d}-passage",
|
|
"text": " ".join(packed),
|
|
"delivery": "narrative"
|
|
if "narrative" in profile.get("deliveries", {})
|
|
else None,
|
|
}
|
|
)
|
|
packed = []
|
|
segments.append(
|
|
{
|
|
"id": f"{len(segments) + 1:02d}-frame",
|
|
"text": block,
|
|
"native_tempo": True,
|
|
}
|
|
)
|
|
continue
|
|
proposed = " ".join([*packed, block])
|
|
if packed and word_count(proposed) > max_words:
|
|
segments.append(
|
|
{
|
|
"id": f"{len(segments) + 1:02d}-passage",
|
|
"text": " ".join(packed),
|
|
"delivery": "narrative"
|
|
if "narrative" in profile.get("deliveries", {})
|
|
else None,
|
|
}
|
|
)
|
|
packed = [block]
|
|
else:
|
|
packed.append(block)
|
|
if packed:
|
|
segments.append(
|
|
{
|
|
"id": f"{len(segments) + 1:02d}-passage",
|
|
"text": " ".join(packed),
|
|
"delivery": "narrative"
|
|
if "narrative" in profile.get("deliveries", {})
|
|
else None,
|
|
}
|
|
)
|
|
for segment in segments:
|
|
if segment.get("delivery") is None:
|
|
segment.pop("delivery", None)
|
|
|
|
for index, segment in enumerate(segments):
|
|
if index:
|
|
segment["context_before"] = last_sentence(
|
|
segments[index - 1]["text"]
|
|
)
|
|
if index + 1 < len(segments):
|
|
segment["context_after"] = first_sentence(
|
|
segments[index + 1]["text"]
|
|
)
|
|
segment.setdefault(
|
|
"pause_after_ms",
|
|
int(profile["assembly"]["default_pause_ms"]),
|
|
)
|
|
plan = {
|
|
"schema_version": "1.0",
|
|
"job_id": args.job_id or slug(args.script.stem),
|
|
"voice": args.voice,
|
|
"output": str(args.output_audio.resolve()),
|
|
"final_transcript_qa": True,
|
|
"segments": segments,
|
|
}
|
|
validate_plan_data(plan)
|
|
json_write(args.output_plan, plan)
|
|
return plan
|
|
|
|
|
|
def resolve_runtime(path: str | None) -> Path:
|
|
return (
|
|
Path(path).expanduser().resolve()
|
|
if path
|
|
else platform_default_runtime().resolve()
|
|
)
|
|
|
|
|
|
def executable(name: str, preferred: str | None = None) -> str:
|
|
if preferred:
|
|
candidate = Path(preferred).expanduser()
|
|
if candidate.exists():
|
|
return str(candidate)
|
|
found = shutil.which(name)
|
|
if not found:
|
|
raise VoiceError(f"Required executable is unavailable: {name}")
|
|
return found
|
|
|
|
|
|
def resolve_asset(relative: str, runtime: Path) -> Path:
|
|
candidate = Path(relative).expanduser()
|
|
if candidate.is_absolute() and candidate.exists():
|
|
return candidate
|
|
for base in (runtime, PLUGIN_ROOT):
|
|
resolved = base / candidate
|
|
if resolved.exists():
|
|
return resolved
|
|
raise VoiceError(f"Missing authorized asset: {relative}")
|
|
|
|
|
|
def select_asset(segment: dict[str, Any], plan: dict[str, Any]) -> str | None:
|
|
if segment.get("fixed_asset"):
|
|
return str(segment["fixed_asset"])
|
|
pool = segment.get("asset_pool")
|
|
if not pool:
|
|
return None
|
|
digest = hashlib.sha256(
|
|
f"{plan['job_id']}:{segment['id']}".encode()
|
|
).digest()
|
|
return str(pool[int.from_bytes(digest[:4], "big") % len(pool)])
|
|
|
|
|
|
def role_for_segment(
|
|
segment: dict[str, Any],
|
|
profile: dict[str, Any],
|
|
) -> tuple[str, dict[str, Any], dict[str, Any]]:
|
|
delivery = profile.get("deliveries", {}).get(segment.get("delivery"), {})
|
|
role_name = segment.get("role") or delivery.get("role") or "default"
|
|
role = profile["roles"][role_name]
|
|
return role_name, role, delivery
|
|
|
|
|
|
def job_paths(plan: dict[str, Any], runtime: Path) -> dict[str, Path]:
|
|
root = runtime / "jobs" / plan["job_id"]
|
|
return {
|
|
"root": root,
|
|
"raw": root / "raw",
|
|
"alignment": root / "alignment",
|
|
"processed": root / "processed",
|
|
"qa": root / "qa",
|
|
}
|
|
|
|
|
|
def import_audio_stack() -> tuple[Any, Any]:
|
|
try:
|
|
import numpy as np
|
|
import soundfile as sf
|
|
except ImportError as exc:
|
|
raise VoiceError(
|
|
"numpy and soundfile must be installed in the Local Voice environment"
|
|
) from exc
|
|
return np, sf
|
|
|
|
|
|
def import_model() -> tuple[Any, Any]:
|
|
try:
|
|
from cosyvoice3 import CosyVoice3, PyDevice
|
|
except ImportError as exc:
|
|
raise VoiceError(
|
|
"cosyvoice3 is not installed in this Python environment"
|
|
) from exc
|
|
return CosyVoice3, PyDevice
|
|
|
|
|
|
def validate_generated(samples: Any, segment_id: str, np: Any) -> None:
|
|
samples = np.asarray(samples).squeeze()
|
|
if samples.ndim != 1 or samples.size == 0:
|
|
raise VoiceError(f"{segment_id}: unexpected audio shape {samples.shape}")
|
|
if not np.isfinite(samples).all():
|
|
raise VoiceError(f"{segment_id}: generated non-finite audio")
|
|
|
|
|
|
def generate_segments(
|
|
plan: dict[str, Any],
|
|
profile: dict[str, Any],
|
|
runtime: Path,
|
|
paths: dict[str, Path],
|
|
*,
|
|
device_name: str,
|
|
resume: bool,
|
|
) -> None:
|
|
np, sf = import_audio_stack()
|
|
CosyVoice3, PyDevice = import_model()
|
|
model_dir = runtime / MODEL_RELATIVE
|
|
if not model_dir.exists():
|
|
raise VoiceError(f"Missing Candle model directory: {model_dir}")
|
|
generated = [
|
|
segment
|
|
for segment in plan["segments"]
|
|
if not select_asset(segment, plan)
|
|
]
|
|
pending = [
|
|
segment
|
|
for segment in generated
|
|
if not (resume and (paths["raw"] / f"{segment['id']}.wav").exists())
|
|
]
|
|
if not pending:
|
|
print("All raw speech segments already exist.")
|
|
return
|
|
print(f"Loading CosyVoice3 on {device_name}...")
|
|
model = CosyVoice3(
|
|
str(model_dir),
|
|
device=PyDevice(device_name),
|
|
use_f16=False,
|
|
)
|
|
for index, segment in enumerate(pending, start=1):
|
|
_, role, _ = role_for_segment(segment, profile)
|
|
reference_audio = resolve_asset(role["reference_audio"], runtime)
|
|
reference_text = resolve_asset(role["reference_transcript"], runtime)
|
|
context = segment.get("context_before")
|
|
if context is None:
|
|
context = role.get(
|
|
"default_context",
|
|
profile["generation"].get("default_context", ""),
|
|
)
|
|
target = str(segment.get("tts_text", segment["text"])).strip()
|
|
spoken = target
|
|
if context:
|
|
spoken = f"{str(context).strip()} ... ... ... {spoken}"
|
|
if segment.get("context_after"):
|
|
spoken += f" ... ... ... {str(segment['context_after']).strip()}"
|
|
prompt_text = PROMPT_PREFIX + reference_text.read_text(
|
|
encoding="utf-8"
|
|
).strip()
|
|
started = time.perf_counter()
|
|
audio = model.inference_zero_shot(
|
|
text=spoken,
|
|
prompt_text=prompt_text,
|
|
prompt_wav=str(reference_audio),
|
|
)
|
|
elapsed = time.perf_counter() - started
|
|
samples = np.asarray(audio, dtype=np.float32).squeeze()
|
|
validate_generated(samples, segment["id"], np)
|
|
destination = paths["raw"] / f"{segment['id']}.wav"
|
|
sf.write(destination, samples, model.sample_rate, subtype="PCM_16")
|
|
duration = samples.size / model.sample_rate
|
|
print(
|
|
f"[{index}/{len(pending)}] {segment['id']}: "
|
|
f"{duration:.2f}s in {elapsed:.2f}s"
|
|
)
|
|
|
|
|
|
def run_whisper(
|
|
source: Path,
|
|
output_dir: Path,
|
|
*,
|
|
whisper_path: str | None,
|
|
) -> Path:
|
|
whisper = executable("whisper", whisper_path)
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
subprocess.run(
|
|
[
|
|
whisper,
|
|
str(source),
|
|
"--model",
|
|
"base",
|
|
"--device",
|
|
"cpu",
|
|
"--fp16",
|
|
"False",
|
|
"--language",
|
|
"en",
|
|
"--word_timestamps",
|
|
"True",
|
|
"--output_format",
|
|
"json",
|
|
"--output_dir",
|
|
str(output_dir),
|
|
"--verbose",
|
|
"False",
|
|
],
|
|
check=True,
|
|
)
|
|
result = output_dir / f"{source.stem}.json"
|
|
if not result.exists():
|
|
raise VoiceError(f"Whisper did not create {result}")
|
|
return result
|
|
|
|
|
|
def align_segments(
|
|
plan: dict[str, Any],
|
|
runtime: Path,
|
|
paths: dict[str, Path],
|
|
*,
|
|
whisper_path: str | None,
|
|
) -> None:
|
|
for segment in plan["segments"]:
|
|
if select_asset(segment, plan):
|
|
continue
|
|
raw = paths["raw"] / f"{segment['id']}.wav"
|
|
if not raw.exists():
|
|
raise VoiceError(f"{segment['id']}: missing raw WAV")
|
|
alignment = paths["alignment"] / f"{segment['id']}.json"
|
|
if not alignment.exists() or raw.stat().st_mtime > alignment.stat().st_mtime:
|
|
run_whisper(
|
|
raw,
|
|
paths["alignment"],
|
|
whisper_path=whisper_path,
|
|
)
|
|
|
|
|
|
def aligned_words(path: Path) -> list[dict[str, Any]]:
|
|
data = json_read(path)
|
|
return [
|
|
word
|
|
for segment in data.get("segments", [])
|
|
for word in segment.get("words", [])
|
|
if normalized_tokens(str(word.get("word", "")))
|
|
]
|
|
|
|
|
|
def locate_phrase(
|
|
phrase: str,
|
|
words: list[dict[str, Any]],
|
|
*,
|
|
start_at: int = 0,
|
|
minimum_score: float = 0.55,
|
|
) -> tuple[int, int, float]:
|
|
expected = normalized_tokens(phrase)[:10]
|
|
if len(expected) < 2:
|
|
raise VoiceError(f"Phrase too short for alignment: {phrase!r}")
|
|
heard = [normalized_tokens(str(word["word"]))[0] for word in words]
|
|
best = (-1.0, -1, -1)
|
|
for start in range(start_at, len(heard)):
|
|
for width in range(max(2, len(expected) - 3), len(expected) + 4):
|
|
candidate = heard[start:start + width]
|
|
if len(candidate) < 2:
|
|
continue
|
|
score = SequenceMatcher(
|
|
None,
|
|
"".join(expected),
|
|
"".join(candidate),
|
|
).ratio()
|
|
if score > best[0]:
|
|
best = (score, start, width)
|
|
if best[0] < minimum_score:
|
|
raise VoiceError(
|
|
f"Could not locate {phrase!r}; confidence={best[0]:.3f}"
|
|
)
|
|
return best[1], best[2], best[0]
|
|
|
|
|
|
def dbfs(value: float) -> float:
|
|
return 20 * math.log10(max(value, 1e-10))
|
|
|
|
|
|
def quiet_cut(
|
|
audio: Any,
|
|
sample_rate: int,
|
|
*,
|
|
lower: float,
|
|
upper: float,
|
|
target: float,
|
|
ceiling: float,
|
|
np: Any,
|
|
) -> tuple[int, float]:
|
|
lo = max(0, round(lower * sample_rate))
|
|
hi = min(len(audio), round(upper * sample_rate))
|
|
window = max(1, round(0.010 * sample_rate))
|
|
if hi - lo < window:
|
|
fallback_lo = max(0, round((upper - 0.14) * sample_rate))
|
|
fallback_hi = min(
|
|
len(audio),
|
|
max(fallback_lo + 1, round((upper - 0.035) * sample_rate)),
|
|
)
|
|
section = np.abs(audio[fallback_lo:fallback_hi])
|
|
chosen = fallback_lo + int(np.argmin(section))
|
|
local = audio[
|
|
max(0, chosen - window // 2):min(len(audio), chosen + window // 2)
|
|
]
|
|
return chosen, dbfs(float(np.sqrt(np.mean(local ** 2))))
|
|
squared = audio[lo:hi] ** 2
|
|
rms = np.sqrt(
|
|
np.convolve(squared, np.ones(window) / window, mode="valid")
|
|
)
|
|
centers = np.arange(len(rms)) + window // 2
|
|
candidates = np.flatnonzero(rms <= 10 ** (ceiling / 20))
|
|
if not candidates.size:
|
|
chosen = int(np.argmin(rms))
|
|
return lo + int(centers[chosen]), dbfs(float(rms[chosen]))
|
|
desired = round(target * sample_rate) - lo
|
|
chosen = candidates[int(np.argmin(np.abs(centers[candidates] - desired)))]
|
|
return lo + int(centers[chosen]), dbfs(float(rms[chosen]))
|
|
|
|
|
|
def transcode_asset(
|
|
source: Path,
|
|
destination: Path,
|
|
*,
|
|
ffmpeg_path: str | None,
|
|
) -> None:
|
|
ffmpeg = executable("ffmpeg", ffmpeg_path)
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
subprocess.run(
|
|
[
|
|
ffmpeg,
|
|
"-hide_banner",
|
|
"-loglevel",
|
|
"error",
|
|
"-y",
|
|
"-i",
|
|
str(source),
|
|
"-filter:a",
|
|
(
|
|
"loudnorm=I=-18:LRA=7:TP=-1.5,"
|
|
"afade=t=in:st=0:d=0.008,"
|
|
"areverse,afade=t=in:st=0:d=0.008,areverse"
|
|
),
|
|
"-ar",
|
|
"24000",
|
|
"-ac",
|
|
"1",
|
|
"-codec:a",
|
|
"pcm_s16le",
|
|
str(destination),
|
|
],
|
|
check=True,
|
|
)
|
|
|
|
|
|
def tempo_for_segment(
|
|
segment: dict[str, Any],
|
|
profile: dict[str, Any],
|
|
role: dict[str, Any],
|
|
delivery: dict[str, Any],
|
|
*,
|
|
body_duration: float,
|
|
target_words: int,
|
|
) -> tuple[float, float | None, float]:
|
|
raw_wpm = target_words / max(body_duration, 0.001) * 60
|
|
short_limit = int(profile["generation"]["short_native_max_words"])
|
|
cadence = profile["cadence"]
|
|
minimum_words = int(cadence["minimum_words_for_normalization"])
|
|
if (
|
|
segment.get("native_tempo")
|
|
or target_words <= short_limit
|
|
or target_words < minimum_words
|
|
):
|
|
return 1.0, None, raw_wpm
|
|
multiplier = segment.get("tempo_multiplier")
|
|
target_wpm = segment.get("target_wpm")
|
|
if multiplier is None and target_wpm is None:
|
|
multiplier = delivery.get("tempo_multiplier")
|
|
target_wpm = delivery.get("target_wpm")
|
|
if multiplier is None and target_wpm is None:
|
|
multiplier = role.get("tempo_multiplier")
|
|
if multiplier is None and target_wpm is None:
|
|
mode = cadence.get("default_mode", "native")
|
|
if mode == "multiplier":
|
|
multiplier = cadence.get("default_tempo_multiplier", 1.0)
|
|
elif mode == "wpm":
|
|
target_wpm = cadence.get("default_wpm")
|
|
if target_wpm is not None:
|
|
tempo = float(target_wpm) / raw_wpm
|
|
else:
|
|
tempo = float(multiplier if multiplier is not None else 1.0)
|
|
minimum = float(cadence["minimum_tempo"])
|
|
maximum = float(cadence["maximum_tempo"])
|
|
if not minimum <= tempo <= maximum:
|
|
raise VoiceError(
|
|
f"{segment['id']}: cadence correction {tempo:.3f} "
|
|
f"outside validated range {minimum:.2f}-{maximum:.2f}"
|
|
)
|
|
return tempo, float(target_wpm) if target_wpm is not None else None, raw_wpm
|
|
|
|
|
|
def apply_processing(
|
|
source: Path,
|
|
destination: Path,
|
|
*,
|
|
tempo: float,
|
|
ffmpeg_path: str | None,
|
|
) -> None:
|
|
ffmpeg = executable("ffmpeg", ffmpeg_path)
|
|
subprocess.run(
|
|
[
|
|
ffmpeg,
|
|
"-hide_banner",
|
|
"-loglevel",
|
|
"error",
|
|
"-y",
|
|
"-i",
|
|
str(source),
|
|
"-filter:a",
|
|
(
|
|
f"atempo={tempo:.6f},"
|
|
"loudnorm=I=-18:LRA=7:TP=-1.5,"
|
|
"afade=t=in:st=0:d=0.008,"
|
|
"areverse,afade=t=in:st=0:d=0.008,areverse"
|
|
),
|
|
"-ar",
|
|
"24000",
|
|
"-ac",
|
|
"1",
|
|
"-codec:a",
|
|
"pcm_s16le",
|
|
str(destination),
|
|
],
|
|
check=True,
|
|
)
|
|
|
|
|
|
def append_suffix(
|
|
body: Any,
|
|
suffix: Any,
|
|
*,
|
|
gap_ms: int,
|
|
sample_rate: int,
|
|
np: Any,
|
|
) -> Any:
|
|
gap = np.zeros(round(gap_ms / 1000 * sample_rate), dtype=np.float32)
|
|
return np.concatenate([body, gap, suffix])
|
|
|
|
|
|
def process_generated_segment(
|
|
segment: dict[str, Any],
|
|
profile: dict[str, Any],
|
|
runtime: Path,
|
|
paths: dict[str, Path],
|
|
*,
|
|
ffmpeg_path: str | None,
|
|
) -> tuple[Any, dict[str, Any]]:
|
|
np, sf = import_audio_stack()
|
|
raw = paths["raw"] / f"{segment['id']}.wav"
|
|
alignment = paths["alignment"] / f"{segment['id']}.json"
|
|
words = aligned_words(alignment)
|
|
target_text = str(segment.get("tts_text", segment["text"]))
|
|
context = segment.get("context_before")
|
|
role_name, role, delivery = role_for_segment(segment, profile)
|
|
if context is None:
|
|
context = role.get(
|
|
"default_context",
|
|
profile["generation"].get("default_context", ""),
|
|
)
|
|
start_index, _, match_confidence = locate_phrase(
|
|
target_text,
|
|
words,
|
|
start_at=2 if context else 0,
|
|
)
|
|
expected_tokens = normalized_tokens(target_text)
|
|
ending_marker = " ".join(expected_tokens[-min(7, len(expected_tokens)):])
|
|
ending_start, ending_width, marker_confidence = locate_phrase(
|
|
ending_marker,
|
|
words,
|
|
start_at=max(start_index, start_index + len(expected_tokens) - 12),
|
|
minimum_score=0.45,
|
|
)
|
|
after_index = min(len(words), ending_start + ending_width)
|
|
after_confidence: float | None = None
|
|
if segment.get("context_after"):
|
|
try:
|
|
context_start, _, after_confidence = locate_phrase(
|
|
str(segment["context_after"]),
|
|
words,
|
|
start_at=after_index,
|
|
minimum_score=0.45,
|
|
)
|
|
after_index = context_start
|
|
except VoiceError:
|
|
after_confidence = None
|
|
target_words = words[start_index:after_index]
|
|
if not target_words:
|
|
raise VoiceError(f"{segment['id']}: alignment selected no target words")
|
|
body_start = float(target_words[0]["start"])
|
|
body_end = float(target_words[-1]["end"])
|
|
audio, sample_rate = sf.read(raw, dtype="float32")
|
|
audio = np.asarray(audio).reshape(-1)
|
|
if start_index == 0:
|
|
start_cut = 0
|
|
start_level = dbfs(
|
|
float(np.sqrt(np.mean(audio[:max(1, sample_rate // 100)] ** 2)))
|
|
)
|
|
else:
|
|
prior_end = float(words[start_index - 1]["end"])
|
|
start_cut, start_level = quiet_cut(
|
|
audio,
|
|
sample_rate,
|
|
lower=prior_end,
|
|
upper=body_start,
|
|
target=body_start - 0.10,
|
|
ceiling=-30.0,
|
|
np=np,
|
|
)
|
|
start_cut = max(start_cut, round(prior_end * sample_rate))
|
|
if after_index == len(words):
|
|
end_cut = len(audio)
|
|
end_level = dbfs(
|
|
float(np.sqrt(np.mean(audio[-max(1, sample_rate // 100):] ** 2)))
|
|
)
|
|
else:
|
|
after_start = float(words[after_index]["start"])
|
|
end_cut, end_level = quiet_cut(
|
|
audio,
|
|
sample_rate,
|
|
lower=body_end + 0.030,
|
|
upper=after_start,
|
|
target=body_end
|
|
+ int(profile["assembly"]["minimum_natural_tail_ms"]) / 1000,
|
|
ceiling=-38.0,
|
|
np=np,
|
|
)
|
|
minimum_end = min(
|
|
len(audio),
|
|
round(
|
|
(
|
|
body_end
|
|
+ min(
|
|
0.08,
|
|
int(profile["assembly"]["minimum_natural_tail_ms"])
|
|
/ 1000,
|
|
)
|
|
)
|
|
* sample_rate
|
|
),
|
|
)
|
|
end_cut = max(end_cut, minimum_end)
|
|
selected = audio[start_cut:end_cut]
|
|
expected = normalized_tokens(target_text)
|
|
heard = [
|
|
normalized_tokens(str(word["word"]))[0]
|
|
for word in target_words
|
|
if normalized_tokens(str(word["word"]))
|
|
]
|
|
matcher = SequenceMatcher(None, expected, heard, autojunk=False)
|
|
matched = sum(block.size for block in matcher.get_matching_blocks())
|
|
coverage = matched / max(1, len(expected))
|
|
ending_expected = expected[-min(7, len(expected)):]
|
|
ending_heard = heard[-min(9, len(heard)):]
|
|
ending = SequenceMatcher(
|
|
None,
|
|
ending_expected,
|
|
ending_heard,
|
|
autojunk=False,
|
|
).ratio()
|
|
if coverage < 0.65 or ending < 0.50:
|
|
raise VoiceError(
|
|
f"{segment['id']}: incomplete target; "
|
|
f"coverage={coverage:.3f}, ending={ending:.3f}"
|
|
)
|
|
tempo, target_wpm, raw_wpm = tempo_for_segment(
|
|
segment,
|
|
profile,
|
|
role,
|
|
delivery,
|
|
body_duration=body_end - body_start,
|
|
target_words=len(expected),
|
|
)
|
|
trimmed = paths["processed"] / f"{segment['id']}-trimmed.wav"
|
|
ready = paths["processed"] / f"{segment['id']}.wav"
|
|
sf.write(trimmed, selected, sample_rate, subtype="PCM_16")
|
|
apply_processing(
|
|
trimmed,
|
|
ready,
|
|
tempo=tempo,
|
|
ffmpeg_path=ffmpeg_path,
|
|
)
|
|
ready_audio, ready_rate = sf.read(ready, dtype="float32")
|
|
if ready_rate != 24000:
|
|
raise VoiceError(f"{segment['id']}: processed sample rate is not 24 kHz")
|
|
result = np.asarray(ready_audio).reshape(-1)
|
|
if segment.get("suffix_asset"):
|
|
suffix_source = resolve_asset(str(segment["suffix_asset"]), runtime)
|
|
suffix_ready = paths["processed"] / f"{segment['id']}-suffix.wav"
|
|
transcode_asset(
|
|
suffix_source,
|
|
suffix_ready,
|
|
ffmpeg_path=ffmpeg_path,
|
|
)
|
|
suffix, suffix_rate = sf.read(suffix_ready, dtype="float32")
|
|
if suffix_rate != 24000:
|
|
raise VoiceError(f"{segment['id']}: suffix sample rate mismatch")
|
|
result = append_suffix(
|
|
result,
|
|
np.asarray(suffix).reshape(-1),
|
|
gap_ms=int(segment.get("suffix_gap_ms", 90)),
|
|
sample_rate=24000,
|
|
np=np,
|
|
)
|
|
sf.write(ready, result, 24000, subtype="PCM_16")
|
|
return result, {
|
|
"id": segment["id"],
|
|
"source": "generated",
|
|
"role": role_name,
|
|
"delivery": segment.get("delivery"),
|
|
"target_match_confidence": round(match_confidence, 3),
|
|
"after_match_confidence": (
|
|
round(after_confidence, 3)
|
|
if after_confidence is not None
|
|
else None
|
|
),
|
|
"transcript_coverage": round(coverage, 3),
|
|
"ending_confidence": round(ending, 3),
|
|
"ending_marker_confidence": round(marker_confidence, 3),
|
|
"retained_pre_word_ms": round(
|
|
(body_start - start_cut / sample_rate) * 1000,
|
|
1,
|
|
),
|
|
"retained_post_word_ms": round(
|
|
(end_cut / sample_rate - body_end) * 1000,
|
|
1,
|
|
),
|
|
"start_cut_dbfs": round(start_level, 1),
|
|
"end_cut_dbfs": round(end_level, 1),
|
|
"raw_wpm": round(raw_wpm, 1),
|
|
"target_wpm": round(target_wpm, 1) if target_wpm else None,
|
|
"applied_tempo": round(tempo, 4),
|
|
"native_tempo": tempo == 1.0,
|
|
"suffix_asset": segment.get("suffix_asset"),
|
|
}
|
|
|
|
|
|
def process_asset_segment(
|
|
segment: dict[str, Any],
|
|
plan: dict[str, Any],
|
|
runtime: Path,
|
|
paths: dict[str, Path],
|
|
*,
|
|
ffmpeg_path: str | None,
|
|
) -> tuple[Any, dict[str, Any]]:
|
|
np, sf = import_audio_stack()
|
|
selected = select_asset(segment, plan)
|
|
if not selected:
|
|
raise VoiceError(f"{segment['id']}: asset selection failed")
|
|
source = resolve_asset(selected, runtime)
|
|
ready = paths["processed"] / f"{segment['id']}.wav"
|
|
transcode_asset(source, ready, ffmpeg_path=ffmpeg_path)
|
|
audio, sample_rate = sf.read(ready, dtype="float32")
|
|
if sample_rate != 24000:
|
|
raise VoiceError(f"{segment['id']}: fixed asset is not 24 kHz")
|
|
return np.asarray(audio).reshape(-1), {
|
|
"id": segment["id"],
|
|
"source": "fixed_asset",
|
|
"asset": str(source),
|
|
"native_tempo": True,
|
|
"duration_seconds": round(len(audio) / sample_rate, 3),
|
|
}
|
|
|
|
|
|
def assemble(
|
|
plan: dict[str, Any],
|
|
profile: dict[str, Any],
|
|
runtime: Path,
|
|
paths: dict[str, Path],
|
|
*,
|
|
ffmpeg_path: str | None,
|
|
whisper_path: str | None,
|
|
final_qa: bool,
|
|
) -> dict[str, Any]:
|
|
np, sf = import_audio_stack()
|
|
processed: list[Any] = []
|
|
reports: list[dict[str, Any]] = []
|
|
for segment in plan["segments"]:
|
|
if select_asset(segment, plan):
|
|
audio, report = process_asset_segment(
|
|
segment,
|
|
plan,
|
|
runtime,
|
|
paths,
|
|
ffmpeg_path=ffmpeg_path,
|
|
)
|
|
else:
|
|
audio, report = process_generated_segment(
|
|
segment,
|
|
profile,
|
|
runtime,
|
|
paths,
|
|
ffmpeg_path=ffmpeg_path,
|
|
)
|
|
processed.append(audio)
|
|
reports.append(report)
|
|
|
|
crossfade = round(
|
|
int(profile["assembly"]["crossfade_ms"]) / 1000 * 24000
|
|
)
|
|
joined = processed[0]
|
|
seam_times: list[float] = []
|
|
for index, right in enumerate(processed[1:], start=1):
|
|
pause_ms = int(
|
|
plan["segments"][index - 1].get(
|
|
"pause_after_ms",
|
|
profile["assembly"]["default_pause_ms"],
|
|
)
|
|
)
|
|
if pause_ms > 0:
|
|
joined = np.concatenate(
|
|
[
|
|
joined,
|
|
np.zeros(round(pause_ms / 1000 * 24000), dtype=np.float32),
|
|
right,
|
|
]
|
|
)
|
|
seam_times.append((len(joined) - len(right)) / 24000)
|
|
else:
|
|
overlap = min(crossfade, len(joined), len(right))
|
|
if overlap:
|
|
theta = np.linspace(
|
|
0,
|
|
np.pi / 2,
|
|
overlap,
|
|
dtype=np.float32,
|
|
)
|
|
seam_times.append((len(joined) - overlap) / 24000)
|
|
joined = np.concatenate(
|
|
[
|
|
joined[:-overlap],
|
|
joined[-overlap:] * np.cos(theta)
|
|
+ right[:overlap] * np.sin(theta),
|
|
right[overlap:],
|
|
]
|
|
)
|
|
else:
|
|
seam_times.append(len(joined) / 24000)
|
|
joined = np.concatenate([joined, right])
|
|
|
|
output = Path(plan["output"])
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
wav_output = output if output.suffix.casefold() == ".wav" else output.with_suffix(
|
|
".wav"
|
|
)
|
|
sf.write(wav_output, joined, 24000, subtype="PCM_16")
|
|
if output.suffix.casefold() == ".mp3":
|
|
ffmpeg = executable("ffmpeg", ffmpeg_path)
|
|
subprocess.run(
|
|
[
|
|
ffmpeg,
|
|
"-hide_banner",
|
|
"-loglevel",
|
|
"error",
|
|
"-y",
|
|
"-i",
|
|
str(wav_output),
|
|
"-codec:a",
|
|
"libmp3lame",
|
|
"-q:a",
|
|
"2",
|
|
str(output),
|
|
],
|
|
check=True,
|
|
)
|
|
|
|
derivative = np.abs(np.diff(joined))
|
|
seam_report = []
|
|
for seam in seam_times:
|
|
center = round(seam * 24000)
|
|
lo = max(0, center - round(0.08 * 24000))
|
|
hi = min(len(derivative), center + round(0.14 * 24000))
|
|
boundary_index = min(max(0, center - 1), len(derivative) - 1)
|
|
boundary_step = (
|
|
float(derivative[boundary_index]) if len(derivative) else 0.0
|
|
)
|
|
seam_report.append(
|
|
{
|
|
"time_seconds": round(seam, 3),
|
|
"boundary_step": round(boundary_step, 6),
|
|
"click_risk": boundary_step > 0.08,
|
|
"maximum_derivative": round(
|
|
float(derivative[lo:hi].max()) if hi > lo else 0.0,
|
|
6,
|
|
),
|
|
}
|
|
)
|
|
|
|
report: dict[str, Any] = {
|
|
"schema_version": "1.0",
|
|
"job_id": plan["job_id"],
|
|
"voice": plan["voice"],
|
|
"output": str(output),
|
|
"wav_output": str(wav_output),
|
|
"duration_seconds": round(len(joined) / 24000, 3),
|
|
"segments": reports,
|
|
"seams": seam_report,
|
|
"global_maximum_derivative": round(float(derivative.max()), 6),
|
|
}
|
|
if final_qa:
|
|
final_alignment = run_whisper(
|
|
output,
|
|
paths["qa"],
|
|
whisper_path=whisper_path,
|
|
)
|
|
data = json_read(final_alignment)
|
|
heard_text = " ".join(
|
|
str(segment.get("text", "")).strip()
|
|
for segment in data.get("segments", [])
|
|
).strip()
|
|
expected = normalized_tokens(
|
|
" ".join(segment["text"] for segment in plan["segments"])
|
|
)
|
|
heard = normalized_tokens(heard_text)
|
|
matcher = SequenceMatcher(None, expected, heard, autojunk=False)
|
|
matched = sum(block.size for block in matcher.get_matching_blocks())
|
|
coverage = matched / max(1, len(expected))
|
|
precision = matched / max(1, len(heard))
|
|
passed = coverage >= 0.60 and precision >= 0.75
|
|
report["final_transcript_qa"] = {
|
|
"coverage": round(coverage, 3),
|
|
"precision": round(precision, 3),
|
|
"heard_text": heard_text,
|
|
"passed": passed,
|
|
}
|
|
if not passed:
|
|
json_write(paths["root"] / "qa-report.json", report)
|
|
raise VoiceError(
|
|
"Final transcript QA failed with "
|
|
f"coverage={coverage:.3f}, precision={precision:.3f}"
|
|
)
|
|
report_path = paths["root"] / "qa-report.json"
|
|
json_write(report_path, report)
|
|
report["report_path"] = str(report_path)
|
|
return report
|
|
|
|
|
|
def doctor(runtime: Path, device_name: str) -> int:
|
|
checks: list[tuple[str, bool, str]] = []
|
|
checks.append(
|
|
(
|
|
"Python",
|
|
sys.version_info >= (3, 10),
|
|
platform.python_version(),
|
|
)
|
|
)
|
|
for name in ("ffmpeg", "whisper"):
|
|
path = shutil.which(name)
|
|
checks.append((name, path is not None, path or "not found"))
|
|
for module in ("numpy", "soundfile", "cosyvoice3"):
|
|
try:
|
|
__import__(module)
|
|
checks.append((module, True, "importable"))
|
|
except ImportError:
|
|
checks.append((module, False, "not importable"))
|
|
model = runtime / MODEL_RELATIVE
|
|
required_model_files = [
|
|
"llm.safetensors",
|
|
"flow.safetensors",
|
|
"hift.safetensors",
|
|
"campplus.onnx",
|
|
"speech_tokenizer_v3.onnx",
|
|
"config.json",
|
|
]
|
|
missing_model = [name for name in required_model_files if not (model / name).exists()]
|
|
checks.append(
|
|
(
|
|
"Candle model",
|
|
not missing_model,
|
|
str(model) if not missing_model else "missing " + ", ".join(missing_model),
|
|
)
|
|
)
|
|
for summary in available_profiles():
|
|
try:
|
|
profile = load_profile(summary["id"])
|
|
missing = []
|
|
for role in profile["roles"].values():
|
|
for field in ("reference_audio", "reference_transcript"):
|
|
try:
|
|
resolve_asset(role[field], runtime)
|
|
except VoiceError:
|
|
missing.append(role[field])
|
|
checks.append(
|
|
(
|
|
summary["display_name"],
|
|
not missing,
|
|
"ready" if not missing else "missing assets: " + ", ".join(missing),
|
|
)
|
|
)
|
|
except VoiceError as exc:
|
|
checks.append((summary["display_name"], False, str(exc)))
|
|
print(f"Runtime: {runtime}")
|
|
print(f"Requested device: {device_name}")
|
|
for name, passed, detail in checks:
|
|
print(f"{'PASS' if passed else 'FAIL'} {name}: {detail}")
|
|
return 0 if all(item[1] for item in checks) else 1
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(prog="local_voice")
|
|
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
|
|
subparsers.add_parser("list", help="List installed voice profiles")
|
|
|
|
validate_parser = subparsers.add_parser(
|
|
"validate",
|
|
help="Validate a render plan without generating audio",
|
|
)
|
|
validate_parser.add_argument("plan", type=Path)
|
|
|
|
plan_parser = subparsers.add_parser(
|
|
"plan",
|
|
help="Create a basic render plan from Markdown",
|
|
)
|
|
plan_parser.add_argument("--voice", required=True)
|
|
plan_parser.add_argument("--script", required=True, type=Path)
|
|
plan_parser.add_argument("--output-plan", required=True, type=Path)
|
|
plan_parser.add_argument("--output-audio", required=True, type=Path)
|
|
plan_parser.add_argument("--job-id")
|
|
|
|
doctor_parser = subparsers.add_parser(
|
|
"doctor",
|
|
help="Check the installed runtime and authorized voice assets",
|
|
)
|
|
doctor_parser.add_argument("--runtime-root")
|
|
doctor_parser.add_argument("--device", default="metal")
|
|
|
|
render_parser = subparsers.add_parser(
|
|
"render",
|
|
help="Generate, align, process, and assemble a render plan",
|
|
)
|
|
render_parser.add_argument("plan", type=Path)
|
|
render_parser.add_argument("--runtime-root")
|
|
render_parser.add_argument(
|
|
"--device",
|
|
choices=("metal", "cpu", "cuda"),
|
|
default="metal" if platform.system() == "Darwin" else "cpu",
|
|
)
|
|
render_parser.add_argument("--resume", action="store_true")
|
|
render_parser.add_argument("--assemble-only", action="store_true")
|
|
render_parser.add_argument("--skip-final-qa", action="store_true")
|
|
render_parser.add_argument("--ffmpeg-path")
|
|
render_parser.add_argument("--whisper-path")
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
if args.command == "list":
|
|
print(json.dumps(available_profiles(), indent=2))
|
|
return 0
|
|
if args.command == "validate":
|
|
plan = json_read(args.plan)
|
|
profile = validate_plan_data(plan)
|
|
print(
|
|
json.dumps(
|
|
{
|
|
"valid": True,
|
|
"voice": profile["id"],
|
|
"segments": len(plan["segments"]),
|
|
"output": plan["output"],
|
|
},
|
|
indent=2,
|
|
)
|
|
)
|
|
return 0
|
|
if args.command == "plan":
|
|
plan = build_plan(args)
|
|
print(json.dumps(plan, indent=2))
|
|
return 0
|
|
if args.command == "doctor":
|
|
return doctor(resolve_runtime(args.runtime_root), args.device)
|
|
if args.command == "render":
|
|
plan = json_read(args.plan)
|
|
profile = validate_plan_data(plan)
|
|
runtime = resolve_runtime(args.runtime_root)
|
|
paths = job_paths(plan, runtime)
|
|
for path in paths.values():
|
|
path.mkdir(parents=True, exist_ok=True)
|
|
if not args.assemble_only:
|
|
generate_segments(
|
|
plan,
|
|
profile,
|
|
runtime,
|
|
paths,
|
|
device_name=args.device,
|
|
resume=args.resume,
|
|
)
|
|
align_segments(
|
|
plan,
|
|
runtime,
|
|
paths,
|
|
whisper_path=args.whisper_path,
|
|
)
|
|
report = assemble(
|
|
plan,
|
|
profile,
|
|
runtime,
|
|
paths,
|
|
ffmpeg_path=args.ffmpeg_path,
|
|
whisper_path=args.whisper_path,
|
|
final_qa=bool(
|
|
plan.get("final_transcript_qa", True)
|
|
and not args.skip_final_qa
|
|
),
|
|
)
|
|
print(json.dumps(report, indent=2))
|
|
return 0
|
|
raise VoiceError(f"Unsupported command: {args.command}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
try:
|
|
raise SystemExit(main())
|
|
except VoiceError as exc:
|
|
print(f"ERROR: {exc}", file=sys.stderr)
|
|
raise SystemExit(2)
|