#!/usr/bin/env python3
"""One-command local speech to text (NVIDIA Parakeet) from GenLovers.

Usage:
    python setup_speech_to_text.py recording.wav

First run: creates a private virtual environment, installs onnx-asr into it, and
downloads the Parakeet TDT 0.6B v3 model (int8) on the first transcription. Later runs
skip both. The transcript is printed and saved next to the audio as a .txt file.
Everything runs on this machine on CPU; nothing is uploaded anywhere. Input is a
WAV file. Needs Python 3.10 or newer (tested on 3.13).
"""
import os
import subprocess
import sys
import venv
from pathlib import Path

ROOT = Path(os.environ.get("GENLOVERS_TOOLS_DIR", Path.home() / "genlovers-tools"))
ENV_DIR = ROOT / "speech-to-text"
BIN = ENV_DIR / ("Scripts" if os.name == "nt" else "bin")
PY = BIN / ("python.exe" if os.name == "nt" else "python")
MODEL = "nemo-parakeet-tdt-0.6b-v3"

# Runs inside the private environment, so onnx-asr is importable there.
RUNNER = """
import sys
import onnx_asr
model = onnx_asr.load_model(sys.argv[1], quantization="int8")
print(model.recognize(sys.argv[2]))
"""


def say(msg):
    print(f"\n==> {msg}", flush=True)


def fail(msg):
    print(f"\nSETUP FAILED: {msg}", file=sys.stderr)
    sys.exit(1)


def ensure_installed():
    if sys.version_info < (3, 10):
        fail("Python 3.10 or newer is required. Install it from https://www.python.org/downloads/")
    if not PY.exists():
        say(f"Creating a private environment in {ENV_DIR}")
        venv.create(ENV_DIR, with_pip=True)
    check = subprocess.run([str(PY), "-c", "import onnx_asr"], capture_output=True)
    if check.returncode != 0:
        say("Installing onnx-asr (this can take a few minutes)...")
        result = subprocess.run([str(PY), "-m", "pip", "install", "--upgrade", "onnx-asr[cpu,hub]"])
        if result.returncode != 0:
            fail("pip could not install onnx-asr. Try Python 3.13, the version this script was tested on, and run this script again.")


def main():
    ensure_installed()
    if len(sys.argv) < 2:
        print(
            "\nSETUP COMPLETE\nNow transcribe a WAV file with:\n"
            "    python setup_speech_to_text.py recording.wav"
        )
        return
    audio = Path(sys.argv[1])
    if not audio.is_file():
        fail(f"{audio} not found.")
    say("Transcribing (the model downloads on the first run)...")
    result = subprocess.run([str(PY), "-c", RUNNER, MODEL, str(audio)], capture_output=True, text=True)
    if result.returncode != 0:
        sys.stderr.write(result.stderr)
        fail("Transcription failed. Make sure the input is a WAV file.")
    transcript = result.stdout.strip()
    out = audio.with_suffix(".txt")
    out.write_text(transcript + "\n", encoding="utf-8")
    print(f"\n{transcript}\n\nSaved {out}")


if __name__ == "__main__":
    main()
