""" Evaluate existing Tatar ASR models on the test split of Common Voice Scripted Speech 27.0 Tatar. Usage: python3 languages/tat/eval.py # all models, full test split python3 languages/tat/eval.py --limit 50 # quick smoke test python3 languages/tat/eval.py --models yasalma/whisper-finetuned-tt-asr python3 languages/tat/eval.py --models omniASR_LLM_1B_v2 python3 languages/tat/eval.py --test-tsv /splits/test.tsv Everything expensive is cached on disk, so reruns are cheap: - the test split -> cv27_tat_test.parquet - each model's output -> results/[__n].tsv With --test-tsv, results go to a results/ folder next to that file instead, so they never mix with results on the official split. Delete a model's TSV to re-run that model. Model ids starting with "omniASR_" are Meta's Omnilingual ASR models and need `pip install omnilingual-asr` (imported lazily, so the rest works without it). Everything else is loaded with the Hugging Face transformers pipeline. Set OMP_NUM_THREADS=1 on cloud machines: NumPy and PyTorch otherwise start a thread per host core, far more than the cores a pod actually gets, and the oversubscription slows feature extraction to a crawl. """ import os import re import unicodedata from pathlib import Path from typing import Annotated, TypedDict import jiwer import librosa import numpy as np import numpy.typing as npt import pandas as pd import torch import typer from datacollective import download_dataset, load_dataset from torch.utils.data import Dataset from tqdm import tqdm from transformers import pipeline COMMON_VOICE_SCRIPTED_SPEECH_27_0_TATAR = "cmu62fysg00noo1077ljif8fv" HERE = Path(__file__).resolve().parent TEST_CACHE = HERE / "cv27_tat_test.parquet" RESULTS_DIR = HERE / "results" SAMPLING_RATE = 16_000 OMNI_LANG = "tat_Cyrl" WORKERS = 1 """ DataLoader processes preparing audio while the GPU transcribes. The ASR pipeline is a ChunkPipeline, which transformers limits to one worker. """ MODELS = [ # Baselines (see the BuzzASR paper, arXiv:2609.09554) "openai/whisper-large-v3", # zero-shot reference for our own fine-tune "omniASR_CTC_1B_v2", "omniASR_LLM_1B_v2", "omniASR_LLM_7B_v2", # needs ~20 GB GPU memory # Tatar fine-tunes on Hugging Face "AigizK/wav2vec2-large-mms-1b-tatar-v2", "yasalma/whisper-finetuned-tt-asr", "AigizK/wav2vec2-large-mms-1b-tatar", "yasalma/whisper-finetuned-tatartts-asr", "anton-l/wav2vec2-large-xlsr-53-tatar", "crang/wav2vec2-large-xlsr-53-tatar", "emre/wav2vec2-large-xlsr-53-W2V2-TATAR-SMALL", "infinitejoy/wav2vec2-large-xls-r-300m-tatar", "kingabzpro/wav2vec2-large-xls-r-300m-Tatar", "sammy786/wav2vec2-xlsr-tatar", # Not tested here # https://github.com/IS2AI/Soyle # https://github.com/IS2AI/TurkicASR ] # Same greedy settings as the BuzzASR paper, so Whisper numbers are comparable. WHISPER_GENERATE_KWARGS = { "language": "tatar", "task": "transcribe", "num_beams": 1, "no_repeat_ngram_size": 3, "repetition_penalty": 1.2, } # -- Types -------------------------------------------------------------------- Audio = npt.NDArray[np.float32] """Mono waveform at SAMPLING_RATE.""" class HFAudioInput(TypedDict): """One input item for the Hugging Face ASR pipeline.""" raw: Audio sampling_rate: int class OmniAudioInput(TypedDict): """One input item for Omnilingual's ASRInferencePipeline.""" waveform: Audio sample_rate: int Row = dict[str, str | float] """One line of the summary table: model id plus scores, or an error message.""" # -- Data --------------------------------------------------------------------- def load_test_split() -> pd.DataFrame: """ Return the CV 27.0 Tatar test split, downloading and caching it on first use. """ if TEST_CACHE.exists(): return pd.read_parquet(TEST_CACHE) if not os.getenv("MDC_API_KEY"): raise RuntimeError( "Set the MDC_API_KEY environment variable to download the dataset." ) download_dataset(COMMON_VOICE_SCRIPTED_SPEECH_27_0_TATAR) dataset = load_dataset(COMMON_VOICE_SCRIPTED_SPEECH_27_0_TATAR) test = dataset[dataset.split == "test"].reset_index(drop=True) test.to_parquet(TEST_CACHE) return test def load_audio(path: str) -> Audio: """ Decode the audio file at `path` to a mono float array at SAMPLING_RATE. """ audio, _ = librosa.load(path, sr=SAMPLING_RATE, mono=True) return audio class AudioDataset(Dataset): """ Clips decoded on access, in the format the HF pipeline expects. Being a Dataset (not a generator) lets the pipeline prepare upcoming clips in a worker process while the GPU transcribes the current batch. """ def __init__(self, paths: list[str]) -> None: self.paths = paths def __len__(self) -> int: return len(self.paths) def __getitem__(self, i: int) -> HFAudioInput: return { "raw": load_audio(self.paths[i]), "sampling_rate": SAMPLING_RATE, } # -- Text normalization ------------------------------------------------------- def normalize(text: str) -> str: """ Lowercase, drop punctuation, collapse whitespace. \\w is Unicode-aware, so ә ө ү җ ң һ are kept. """ text = unicodedata.normalize("NFC", text).lower() text = re.sub(r"[^\w\s]|_", " ", text) return " ".join(text.split()) # -- Inference --------------------------------------------------------------- def free_gpu() -> None: """Release cached GPU memory so the next model has room to load.""" if torch.cuda.is_available(): torch.cuda.empty_cache() def transcribe_hf( model_id: str, paths: list[str], batch_size: int ) -> list[str]: """Transcribe `paths` with a Hugging Face model, one hypothesis per path.""" use_cuda = torch.cuda.is_available() asr = pipeline( "automatic-speech-recognition", model=model_id, device="cuda:0" if use_cuda else "cpu", dtype=torch.float16 if use_cuda else torch.float32, ) inputs = AudioDataset(paths) if "whisper" in model_id.lower(): outputs = asr( inputs, batch_size=batch_size, num_workers=WORKERS, generate_kwargs=WHISPER_GENERATE_KWARGS, ) else: outputs = asr(inputs, batch_size=batch_size, num_workers=WORKERS) hyps = [ out["text"] for out in tqdm(outputs, total=len(paths), desc=model_id) ] del asr free_gpu() return hyps def transcribe_omnilingual( model_card: str, paths: list[str], batch_size: int ) -> list[str]: """ Transcribe `paths` with a Meta Omnilingual ASR model, one hypothesis per path. """ from omnilingual_asr.models.inference.pipeline import ASRInferencePipeline try: from omnilingual_asr.models.wav2vec2_llama.lang_ids import ( supported_langs, ) assert OMNI_LANG in supported_langs, ( f"{OMNI_LANG} not supported by Omnilingual" ) except ImportError: pass asr = ASRInferencePipeline(model_card=model_card) use_lang = ( "_LLM_" in model_card ) # CTC models have no language conditioning? hyps: list[str] = [] for i in tqdm(range(0, len(paths), batch_size), desc=model_card): chunk = paths[i : i + batch_size] audio: list[OmniAudioInput] = [ {"waveform": load_audio(p), "sample_rate": SAMPLING_RATE} for p in chunk ] kwargs = {"lang": [OMNI_LANG] * len(chunk)} if use_lang else {} hyps.extend(asr.transcribe(audio, batch_size=len(chunk), **kwargs)) del asr free_gpu() return hyps def transcribe(model_id: str, paths: list[str], batch_size: int) -> list[str]: """Transcribe `paths` with `model_id`, dispatching to the right backend.""" if model_id.startswith("omniASR"): return transcribe_omnilingual(model_id, paths, batch_size) return transcribe_hf(model_id, paths, batch_size) def load_test_tsv(path: Path) -> pd.DataFrame: """Return a test set saved by finetune.py (its splits/test.tsv).""" return pd.read_csv(path, sep="\t", keep_default_na=False) def results_dir_for(test_tsv: Path | None) -> Path: """ Return where results go: RESULTS_DIR for the official split, else a results/ folder next to `test_tsv`. """ return RESULTS_DIR if test_tsv is None else test_tsv.parent / "results" def predictions_path( model_id: str, limit: int, results_dir: Path | None = None ) -> Path: """ Return the cache file for `model_id`'s predictions on the first `limit` clips, in `results_dir` (default RESULTS_DIR). """ name = model_id.replace("/", "__") suffix = f"__n{limit}" if limit else "" return (results_dir or RESULTS_DIR) / f"{name}{suffix}.tsv" def get_predictions( model_id: str, test: pd.DataFrame, limit: int, batch_size: int, results_dir: Path | None = None, ) -> pd.DataFrame: """ Return reference/hypothesis pairs for `model_id` on `test`, from cache if present. """ path = predictions_path(model_id, limit, results_dir) if path.exists(): return pd.read_csv(path, sep="\t", keep_default_na=False) hyps = transcribe(model_id, list(test.audio_path), batch_size) df = pd.DataFrame( { "audio_path": test.audio_path, "reference": test.transcription, "hypothesis": hyps, } ) path.parent.mkdir(parents=True, exist_ok=True) df.to_csv(path, sep="\t", index=False) return df # -- Scoring ------------------------------------------------------------------ def score(df: pd.DataFrame) -> dict[str, float]: """ Compute WER and CER of `df.hypothesis` against `df.reference`, normalized and raw. """ refs, hyps = list(df.reference), list(df.hypothesis) refs_n, hyps_n = [normalize(t) for t in refs], [normalize(t) for t in hyps] return { "wer": jiwer.wer(refs_n, hyps_n), "cer": jiwer.cer(refs_n, hyps_n), "wer_raw": jiwer.wer(refs, hyps), "cer_raw": jiwer.cer(refs, hyps), } def evaluate( model_id: str, test: pd.DataFrame, limit: int, batch_size: int, results_dir: Path | None = None, ) -> tuple[pd.DataFrame | None, Row]: """ One model -> (predictions, summary row). Errors become a row, not a crash. """ try: preds = get_predictions(model_id, test, limit, batch_size, results_dir) return preds, {"model": model_id, **score(preds)} except Exception as e: print(f"!! {model_id} failed: {type(e).__name__}: {e}") return None, {"model": model_id, "error": f"{type(e).__name__}: {e}"} def summarize( rows: list[Row], limit: int, results_dir: Path | None = None ) -> pd.DataFrame: """ Combine per-model rows into a table sorted by CER, and save it as TSV in `results_dir` (default RESULTS_DIR). """ summary = pd.DataFrame(rows) if "cer" in summary: summary = summary.sort_values("cer", na_position="last") directory = results_dir or RESULTS_DIR directory.mkdir(parents=True, exist_ok=True) name = f"summary__n{limit}.tsv" if limit else "summary.tsv" summary.to_csv(directory / name, sep="\t", index=False) return summary # ── Main ────────────────────────────────────────────────────────────────── def main( models: Annotated[ list[str], typer.Option(help="Model ids to evaluate (repeat the flag)."), ] = MODELS, limit: Annotated[ int, typer.Option( help="Evaluate on the first N test clips only (0 = all)." ), ] = 0, batch_size: Annotated[ int, typer.Option(help="Inference batch size.") ] = 16, test_tsv: Annotated[ Path | None, typer.Option(help="Test set saved by finetune.py (splits/test.tsv)."), ] = None, ) -> tuple[pd.DataFrame, dict[str, pd.DataFrame], pd.DataFrame]: """Evaluate Tatar ASR models on the CV 27.0 Tatar test split.""" test = load_test_split() if test_tsv is None else load_test_tsv(test_tsv) results_dir = results_dir_for(test_tsv) if limit: test = test.head(limit) print(f"Test clips: {len(test)}, speakers: {test.speaker_id.nunique()}") results = { m: evaluate(m, test, limit, batch_size, results_dir) for m in models } preds = {m: p for m, (p, _) in results.items() if p is not None} summary = summarize( [row for _, row in results.values()], limit, results_dir ) with pd.option_context( "display.max_colwidth", 60, "display.float_format", "{:.3f}".format ): print(summary.to_string(index=False)) return test, preds, summary if __name__ == "__main__": typer.run(main)