diff options
Diffstat (limited to 'tools')
| -rwxr-xr-x | tools/download-parakeet.sh | 33 | ||||
| -rw-r--r-- | tools/whistle-eval/.gitignore | 7 | ||||
| -rw-r--r-- | tools/whistle-eval/README.md | 94 | ||||
| -rw-r--r-- | tools/whistle-eval/build-data.mjs | 19 | ||||
| -rw-r--r-- | tools/whistle-eval/evaluate.py | 141 | ||||
| -rw-r--r-- | tools/whistle-eval/package-lock.json | 229 | ||||
| -rw-r--r-- | tools/whistle-eval/package.json | 14 | ||||
| -rw-r--r-- | tools/whistle-eval/parakeet_baseline.py | 67 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/README.md | 41 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/en1.txt | 1 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/en2.txt | 1 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/it1.txt | 1 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/it1_noisy.txt | 1 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/it2.txt | 1 | ||||
| -rw-r--r-- | tools/whistle-eval/samples/jfk.txt | 1 | ||||
| -rwxr-xr-x | tools/whistle-eval/setup.sh | 38 | ||||
| -rw-r--r-- | tools/whistle-eval/whisper_baseline.py | 65 |
17 files changed, 754 insertions, 0 deletions
diff --git a/tools/download-parakeet.sh b/tools/download-parakeet.sh new file mode 100755 index 0000000..cb6fafc --- /dev/null +++ b/tools/download-parakeet.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +# Downloads NVIDIA Parakeet TDT 0.6B v3 (int8) plus the Silero VAD model into a +# folder whose layout matches what RECCoon expects on the device. +# +# Usage: +# ./download-parakeet.sh [output-dir] (default: ./reccoon-models) +# +# Then copy the output directory to the device, e.g.: +# adb push reccoon-models/ /sdcard/Android/data/com.wuhei.reccoon/files/models/ +set -euo pipefail + +OUT="${1:-reccoon-models}" +PARAKEET="https://huggingface.co/csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v3-int8/resolve/main" +VAD="https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/silero_vad.onnx" + +mkdir -p "$OUT/parakeet-v3" + +echo "==> Silero VAD" +curl -L --fail -o "$OUT/silero_vad.onnx" "$VAD" + +echo "==> Parakeet v3 (encoder is ~652 MB, this can take a while)" +for file in encoder.int8.onnx decoder.int8.onnx joiner.int8.onnx tokens.txt; do + echo " -> $file" + curl -L --fail -o "$OUT/parakeet-v3/$file" "$PARAKEET/$file" +done + +echo +echo "Done. Layout:" +find "$OUT" -type f -printf ' %p (%s bytes)\n' +echo +echo "On the phone these must live under:" +echo " /sdcard/Android/data/com.wuhei.reccoon/files/models/" +echo "i.e. .../models/silero_vad.onnx and .../models/parakeet-v3/*.onnx" diff --git a/tools/whistle-eval/.gitignore b/tools/whistle-eval/.gitignore new file mode 100644 index 0000000..f28fcce --- /dev/null +++ b/tools/whistle-eval/.gitignore @@ -0,0 +1,7 @@ +node_modules/ +vendor/ +models/ +.venv/ +__pycache__/ +samples/*.wav +samples/*.mp3 diff --git a/tools/whistle-eval/README.md b/tools/whistle-eval/README.md new file mode 100644 index 0000000..ed6e71e --- /dev/null +++ b/tools/whistle-eval/README.md @@ -0,0 +1,94 @@ +# Speech-model evaluation + +Compares on-device speech recognition candidates on the same 16 kHz mono clips: + +- the multilingual Whisper tiny/base models that RECCoon already uses, + through the exact sherpa-onnx 1.13.8 version the Android app bundles; +- [NVIDIA Parakeet TDT 0.6B v3](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) + (25 European languages) via its sherpa-onnx int8 export; +- [Cactus Whistle](https://huggingface.co/mrfakename/whistle-ONNX), a custom + *Needle* encoder/decoder run through its reference Node pipeline. + +Whistle is **not** a Whisper model and cannot be loaded with sherpa-onnx. +Parakeet, unlike Whistle, **is** supported by sherpa-onnx. + +## Setup + +Requires Node 18+, Python 3, `ffmpeg`, and `curl`. + +```sh +cd tools/whistle-eval + +# 1. Whistle pack + ONNX graphs + reference JS + onnxruntime-node +# and the Whisper tiny/base int8 models. +./setup.sh + +# 2. Parakeet v3 int8 (~660 MB, optional) +mkdir -p models/parakeet-v3-int8 +BASE=https://huggingface.co/csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v3-int8/resolve/main +for f in encoder.int8.onnx decoder.int8.onnx joiner.int8.onnx tokens.txt; do + curl -L --fail -o "models/parakeet-v3-int8/$f" "$BASE/$f" +done + +# 3. Python environment for the sherpa-onnx baselines. +python3 -m venv .venv +.venv/bin/pip install sherpa-onnx numpy +``` + +`setup.sh` downloads the 17 MB `whistle.pack` and rebuilds the fp32 +`encoder.onnx.data` / `decoder.onnx.data` files locally with `build-data.mjs`. +They are **not** committed (see `.gitignore`); the pack is enough to recreate +them byte-exactly. + +## Run + +```sh +# One clip (language inferred: `it*` -> Italian, otherwise English) +.venv/bin/python evaluate.py samples/jfk.wav + +# A whole folder +.venv/bin/python evaluate.py "samples/*.wav" +``` + +`evaluate.py` prints the transcript and the word error rate (WER) against +`samples/<name>.txt` when a reference exists. Parakeet is included +automatically when `models/parakeet-v3-int8/` is present. + +| clip | Whisper tiny | Whisper base | Whistle | **Parakeet v3** | +|---|---|---|---|---| +| `jfk.wav` | 0% | 4.5% | 0% | 0% | +| `en1.wav` | 0% | 0% | 8.7% | 0% | +| `en2.wav` | 13.6% | 13.6% | 13.6% | 13.6% | +| `it1.wav` | 24% | 12% | 20% | **4%** | +| `it1_noisy.wav` | 24% | 12% | 24% | **0%** | +| `it2.wav` | 21.7% | 26.1% | 4.3% | 4.3% | + +Average WER over these six clips (lower is better): + +| model | English (3) | Italian (3) | All (6) | int8 size | +|---|---|---|---|---| +| Whisper tiny | 4.5% | 23.2% | 13.9% | ~103 MB | +| Whisper base | 6.0% | 16.7% | 11.4% | ~160 MB | +| Whistle | 7.4% | 16.1% | 11.8% | ~17 MB pack (desktop) | +| **Parakeet v3** | 4.5% | **2.8%** | **3.7%** | ~670 MB | + +### Reading the result + +- **Parakeet v3 is the clear winner**, especially for Italian and noisy audio: + it was perfect on the noisy Italian clip that fooled every other model. +- The only Parakeet "errors" are number formatting ("four fifteen" → "4:15"), + which WER counts even though the transcript is correct. +- The cost is size: ~670 MB int8, about four times Whisper base. It is usable + on a modern phone but is a large download and a lot of RAM. +- Whisper base stays the best small model; Whistle is competitive but not a + clear win, and needs a custom runtime. +- These are small neural-TTS clips, not real recordings. Run the harness on + real RECCoon recordings before tuning further. + +## Android notes + +- Parakeet can be added through the existing sherpa-onnx AAR with + `model_type="nemo_transducer"` (Java: `OfflineTransducerModelConfig`). RECCoon + 1.5.0-alpha exposes it as the "Parakeet v3 (best)" model in the player. +- Whistle would need `onnxruntime-android` plus the log-mel frontend, BPE + tokenizer, engram features and greedy decoder from `vendor/js/whistle.js`. diff --git a/tools/whistle-eval/build-data.mjs b/tools/whistle-eval/build-data.mjs new file mode 100644 index 0000000..20be062 --- /dev/null +++ b/tools/whistle-eval/build-data.mjs @@ -0,0 +1,19 @@ +// Rebuilds the fp32 ONNX external-data files from the compact whistle.pack. +// This is the "small weights" path: instead of committing ~145 MB of .onnx.data +// to git (or downloading it), keep the 17 MB pack and expand it on first run. +import fs from 'fs'; +import { parsePack, buildModelData } from './vendor/js/pack.js'; + +const root = new URL('./vendor/', import.meta.url); +const packBytes = fs.readFileSync(new URL('whistle.pack', root)); +const pack = parsePack( + packBytes.buffer.slice(packBytes.byteOffset, packBytes.byteOffset + packBytes.byteLength)); + +for (const name of ['encoder', 'decoder']) { + const target = new URL(`onnx/${name}.onnx.data`, root); + const data = buildModelData(pack, name, (p) => { + process.stdout.write(`\r${name}: ${(p * 100).toFixed(0)}% `); + }); + fs.writeFileSync(target, data); + process.stdout.write(`\r${name}: wrote ${data.length} bytes\n`); +} diff --git a/tools/whistle-eval/evaluate.py b/tools/whistle-eval/evaluate.py new file mode 100644 index 0000000..f4437ea --- /dev/null +++ b/tools/whistle-eval/evaluate.py @@ -0,0 +1,141 @@ +#!/usr/bin/env python3 +"""Compare Cactus Whistle with the app's Whisper models on the same clips. + +Usage: + python evaluate.py samples/jfk.wav en + python evaluate.py samples/*.wav # language guessed from a *.it.wav suffix + +For every clip this runs: + * Whisper tiny (int8, multilingual) through sherpa-onnx + * Whisper base (int8, multilingual) through sherpa-onnx + * Cactus Whistle through its reference Node pipeline +and prints the transcript plus the word error rate (WER) against +`samples/<name>.txt` when that reference file exists. + +Setup (see README.md): + npm install + node build-data.mjs + python3 -m venv .venv && .venv/bin/pip install sherpa-onnx numpy +""" +import argparse +import glob +import os +import re +import subprocess +import sys +import unicodedata +import wave + +HERE = os.path.dirname(os.path.abspath(__file__)) +VENV = os.environ.get("WHISTLE_EVAL_PYTHON", sys.executable) + + +def normalize(text: str) -> list: + text = unicodedata.normalize("NFKD", text.lower()) + text = "".join(c for c in text if not unicodedata.combining(c)) + text = re.sub(r"[^a-z0-9]+", " ", text) + return text.split() + + +def wer(reference: list, hypothesis: list) -> float: + if not reference: + return 0.0 if not hypothesis else 1.0 + previous = list(range(len(hypothesis) + 1)) + for i, ref in enumerate(reference, start=1): + current = [i] + for j, hyp in enumerate(hypothesis, start=1): + cost = 0 if ref == hyp else 1 + current.append(min(previous[j] + 1, current[j - 1] + 1, previous[j - 1] + cost)) + previous = current + return previous[-1] / len(reference) + + +def read_reference(audio: str): + base = os.path.splitext(audio)[0] + for candidate in (base + ".txt", audio + ".txt"): + if os.path.isfile(candidate): + with open(candidate, encoding="utf-8") as handle: + return handle.read().strip() + return None + + +def run_whistle(audio: str) -> str: + out = subprocess.run( + ["node", os.path.join(HERE, "vendor", "js", "example.mjs"), audio], + cwd=HERE, capture_output=True, text=True, timeout=600) + if out.returncode != 0: + return f"<whistle failed: {out.stderr.strip().splitlines()[-1] if out.stderr else '?'}>" + first = out.stdout.strip().splitlines()[0] if out.stdout.strip() else "" + return re.sub(r"^\[[a-z]{2}\]\s*", "", first) + + +def parakeet_available() -> bool: + return os.path.isfile(os.path.join( + HERE, "models", "parakeet-v3-int8", "encoder.int8.onnx")) + + +def run_parakeet(audio: str) -> str: + script = os.path.join(HERE, "parakeet_baseline.py") + out = subprocess.run( + [VENV, script, audio], capture_output=True, text=True, timeout=1200) + if out.returncode != 0: + tail = out.stderr.strip().splitlines() + return f"<parakeet failed: {tail[-1] if tail else '?'}>" + return out.stdout.strip() + + +def run_whisper(audio: str, size: str, language: str) -> str: + script = os.path.join(HERE, "whisper_baseline.py") + out = subprocess.run( + [VENV, script, audio, size, language], + capture_output=True, text=True, timeout=900) + if out.returncode != 0: + tail = out.stderr.strip().splitlines() + return f"<whisper {size} failed: {tail[-1] if tail else '?'}>" + return out.stdout.strip() + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("clips", nargs="+") + parser.add_argument("language", nargs="?", default=None) + args = parser.parse_args() + + clips = [] + for pattern in args.clips: + clips.extend(sorted(glob.glob(pattern)) or [pattern]) + + print(f"{'clip':<16} {'model':<14} {'WER':>7} transcript") + print("-" * 100) + for audio in clips: + if not os.path.isfile(audio): + print(f"{audio}: not found") + continue + language = args.language + name = os.path.basename(audio).lower() + if language is None: + language = "it" if (name.startswith("it") or ".it." in name) else "en" + reference = read_reference(audio) + ref_words = normalize(reference) if reference else None + + results = [ + ("whistle", run_whistle(audio)), + ("whisper tiny", run_whisper(audio, "tiny", language)), + ("whisper base", run_whisper(audio, "base", language)), + ] + if parakeet_available(): + results.append(("parakeet v3", run_parakeet(audio))) + for model, text in results: + if ref_words is None: + score = " n/a" + else: + score = f"{wer(ref_words, normalize(text)) * 100:6.1f}%" + print(f"{os.path.basename(audio):<16} {model:<14} {score} {text}") + if reference: + print(f"{'':<16} {'reference':<14} {'':>7} {reference}") + print() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/whistle-eval/package-lock.json b/tools/whistle-eval/package-lock.json new file mode 100644 index 0000000..063b6cc --- /dev/null +++ b/tools/whistle-eval/package-lock.json @@ -0,0 +1,229 @@ +{ + "name": "reccoon-whistle-eval", + "version": "1.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "reccoon-whistle-eval", + "version": "1.0.0", + "dependencies": { + "onnxruntime-node": "^1.20.1" + } + }, + "node_modules/adm-zip": { + "version": "0.6.1", + "resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.6.1.tgz", + "integrity": "sha512-Xwrja8nx9e5o2N1my4DsKCeKpdrnACyr1wtbPxBDgGzKzKyE9kRtBFA8mWldI+RVlD7CBZNWY/wQ2+ydwOR6kQ==", + "license": "MIT", + "engines": { + "node": ">=14.0" + } + }, + "node_modules/define-data-property": { + "version": "1.1.4", + "resolved": "https://registry.npmjs.org/define-data-property/-/define-data-property-1.1.4.tgz", + "integrity": "sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==", + "license": "MIT", + "dependencies": { + "es-define-property": "^1.0.0", + "es-errors": "^1.3.0", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/define-properties": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/define-properties/-/define-properties-1.2.1.tgz", + "integrity": "sha512-8QmQKqEASLd5nx0U1B1okLElbUuuttJ/AnYmRXbbbGDWh6uS208EjD4Xqq/I9wK7u0v6O08XhTWnt5XtEbR6Dg==", + "license": "MIT", + "dependencies": { + "define-data-property": "^1.0.1", + "has-property-descriptors": "^1.0.0", + "object-keys": "^1.1.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/es-define-property": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/es-define-property/-/es-define-property-1.0.1.tgz", + "integrity": "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g==", + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/es-errors": { + "version": "1.3.0", + "resolved": "https://registry.npmjs.org/es-errors/-/es-errors-1.3.0.tgz", + "integrity": "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw==", + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/escape-string-regexp": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz", + "integrity": "sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA==", + "license": "MIT", + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/global-agent": { + "version": "4.1.3", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-4.1.3.tgz", + "integrity": "sha512-KUJEViiuFT3I97t+GYMikLPJS2Lfo/S2F+DQuBWzuzaMPnvt5yyZePzArx36fBzpGTxZjIpDbXLeySLgh+k76g==", + "license": "BSD-3-Clause", + "dependencies": { + "globalthis": "^1.0.2", + "matcher": "^4.0.0", + "semver": "^7.3.5", + "serialize-error": "^8.1.0" + }, + "engines": { + "node": ">=10.0" + } + }, + "node_modules/globalthis": { + "version": "1.0.4", + "resolved": "https://registry.npmjs.org/globalthis/-/globalthis-1.0.4.tgz", + "integrity": "sha512-DpLKbNU4WylpxJykQujfCcwYWiV/Jhm50Goo0wrVILAv5jOr9d+H+UR3PhSCD2rCCEIg0uc+G+muBTwD54JhDQ==", + "license": "MIT", + "dependencies": { + "define-properties": "^1.2.1", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/gopd": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz", + "integrity": "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg==", + "license": "MIT", + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/has-property-descriptors": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/has-property-descriptors/-/has-property-descriptors-1.0.2.tgz", + "integrity": "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg==", + "license": "MIT", + "dependencies": { + "es-define-property": "^1.0.0" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/matcher": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-4.0.0.tgz", + "integrity": "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ==", + "license": "MIT", + "dependencies": { + "escape-string-regexp": "^4.0.0" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/object-keys": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/object-keys/-/object-keys-1.1.1.tgz", + "integrity": "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA==", + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/onnxruntime-common": { + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.30.0.tgz", + "integrity": "sha512-7fdVWjAID1dVhH/G8qK3APARunV4VkBFoCQAP7qp4Wkab0mrorvmc+sqiT+mKXOzDqdjN5j+/Z9nb4gzNPWcyA==", + "license": "MIT" + }, + "node_modules/onnxruntime-node": { + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.30.0.tgz", + "integrity": "sha512-twhs1C2C/BFkz1yc5OY0KIU2GUq6DURO7hD4bx5Q2Qy3nAMJwRXW8xU3NVczE29VA9lolLOYepoD8fjTGOfIqw==", + "hasInstallScript": true, + "license": "MIT", + "os": [ + "win32", + "darwin", + "linux" + ], + "dependencies": { + "adm-zip": "^0.6.0", + "global-agent": "^4.1.3", + "onnxruntime-common": "1.30.0" + } + }, + "node_modules/semver": { + "version": "7.8.5", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", + "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", + "license": "ISC", + "bin": { + "semver": "bin/semver.js" + }, + "engines": { + "node": ">=10" + } + }, + "node_modules/serialize-error": { + "version": "8.1.0", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-8.1.0.tgz", + "integrity": "sha512-3NnuWfM6vBYoy5gZFvHiYsVbafvI9vZv/+jlIigFn4oP4zjNPK3LhcY0xSCgeb1a5L8jO71Mit9LlNoi2UfDDQ==", + "license": "MIT", + "dependencies": { + "type-fest": "^0.20.2" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/type-fest": { + "version": "0.20.2", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.20.2.tgz", + "integrity": "sha512-Ne+eE4r0/iWnpAxD852z3A+N0Bt5RN//NjJwRd2VFHEmrywxf5vsZlh4R6lixl6B+wz/8d+maTSAkN1FIkI3LQ==", + "license": "(MIT OR CC0-1.0)", + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + } + } +} diff --git a/tools/whistle-eval/package.json b/tools/whistle-eval/package.json new file mode 100644 index 0000000..b34444b --- /dev/null +++ b/tools/whistle-eval/package.json @@ -0,0 +1,14 @@ +{ + "name": "reccoon-whistle-eval", + "version": "1.0.0", + "private": true, + "type": "module", + "description": "Desktop evaluation harness for the Cactus Whistle ASR model.", + "scripts": { + "build-data": "node build-data.mjs", + "transcribe": "node vendor/js/example.mjs" + }, + "dependencies": { + "onnxruntime-node": "^1.20.1" + } +} diff --git a/tools/whistle-eval/parakeet_baseline.py b/tools/whistle-eval/parakeet_baseline.py new file mode 100644 index 0000000..16987de --- /dev/null +++ b/tools/whistle-eval/parakeet_baseline.py @@ -0,0 +1,67 @@ +#!/usr/bin/env python3 +"""Run NVIDIA Parakeet TDT 0.6B v3 (multilingual) through sherpa-onnx. + +Usage: + python parakeet_baseline.py <audio.wav> + +Requires a virtualenv with `sherpa-onnx` installed and the int8 model under +./models/parakeet-v3-int8/ (see README.md). +""" +import argparse +import os +import sys +import wave + +import numpy as np +import sherpa_onnx + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("audio") + args = parser.parse_args() + + model_dir = os.path.join(os.path.dirname(__file__), "models", "parakeet-v3-int8") + encoder = os.path.join(model_dir, "encoder.int8.onnx") + decoder = os.path.join(model_dir, "decoder.int8.onnx") + joiner = os.path.join(model_dir, "joiner.int8.onnx") + tokens = os.path.join(model_dir, "tokens.txt") + for path in (encoder, decoder, joiner, tokens): + if not os.path.isfile(path): + print(f"missing {path}", file=sys.stderr) + return 2 + + recognizer = sherpa_onnx.OfflineRecognizer.from_transducer( + encoder=encoder, + decoder=decoder, + joiner=joiner, + tokens=tokens, + num_threads=2, + sample_rate=16000, + feature_dim=80, + decoding_method="greedy_search", + model_type="nemo_transducer", + debug=False, + ) + + with wave.open(args.audio, "rb") as wav: + sample_rate = wav.getframerate() + channels = wav.getnchannels() + width = wav.getsampwidth() + frames = wav.readframes(wav.getnframes()) + if width != 2: + print("only 16-bit PCM WAV is supported", file=sys.stderr) + return 2 + samples = np.frombuffer(frames, dtype=np.int16).astype(np.float32) / 32768.0 + if channels > 1: + samples = samples.reshape(-1, channels).mean(axis=1) + + stream = recognizer.create_stream() + stream.accept_waveform(sample_rate, samples) + recognizer.decode_stream(stream) + print(stream.result.text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/whistle-eval/samples/README.md b/tools/whistle-eval/samples/README.md new file mode 100644 index 0000000..28f0d12 --- /dev/null +++ b/tools/whistle-eval/samples/README.md @@ -0,0 +1,41 @@ +# Test clips + +The audio files are not committed. Recreate them with `edge-tts` (neural TTS) +and `ffmpeg`, or drop your own 16 kHz mono 16-bit WAV files here. When a +`<name>.txt` sidecar exists, `evaluate.py` reports the WER against it. + +```sh +# English +edge-tts --voice en-US-AriaNeural \ + --text "The quick brown fox jumps over the lazy dog near the riverbank, while the scientists carefully measured the temperature of the experimental apparatus." \ + --write-media en1.mp3 +# Italian +edge-tts --voice it-IT-ElsaNeural \ + --text "Il rapido volpe marrone salta sopra il cane pigro vicino alla riva del fiume, mentre gli scienziati misurano con attenzione la temperatura dell'apparecchiatura sperimentale." \ + --write-media it1.mp3 +# ...then: +for f in en1 it1; do + ffmpeg -y -i "$f.mp3" -ar 16000 -ac 1 -c:a pcm_s16le "$f.wav" +done +``` + +The JFK clip is the public sample from +[whisper.cpp](https://github.com/ggerganov/whisper.cpp/blob/master/samples/jfk.wav). + +## References used by the committed test set + +| file | reference | +|---|---| +| `jfk.wav` | And so my fellow Americans ask not what your country can do for you ask what you can do for your country | +| `en1.wav` | The quick brown fox jumps over the lazy dog near the riverbank, while the scientists carefully measured the temperature of the experimental apparatus. | +| `en2.wav` | Please call me back at four fifteen in the afternoon. The meeting is with Doctor Smith and Professor Johnson from the university. | +| `it1.wav` | Il rapido volpe marrone salta sopra il cane pigro vicino alla riva del fiume, mentre gli scienziati misurano con attenzione la temperatura dell'apparecchiatura sperimentale. | +| `it2.wav` | Per favore, richiamami alle quattro e un quarto del pomeriggio. L'incontro e con il dottor Rossi e la professoressa Bianchi dell'universita. | + +`it1_noisy.wav` is `it1.wav` mixed with low-level white noise: + +```sh +ffmpeg -y -i it1.wav -f lavfi -i "anoisesrc=color=white:amplitude=0.03:duration=999" \ + -filter_complex "[0:a][1:a]amix=inputs=2:duration=first:weights=1 0.35" \ + -ar 16000 -ac 1 -c:a pcm_s16le it1_noisy.wav +``` diff --git a/tools/whistle-eval/samples/en1.txt b/tools/whistle-eval/samples/en1.txt new file mode 100644 index 0000000..2848fe3 --- /dev/null +++ b/tools/whistle-eval/samples/en1.txt @@ -0,0 +1 @@ +The quick brown fox jumps over the lazy dog near the riverbank, while the scientists carefully measured the temperature of the experimental apparatus. diff --git a/tools/whistle-eval/samples/en2.txt b/tools/whistle-eval/samples/en2.txt new file mode 100644 index 0000000..da8a6ff --- /dev/null +++ b/tools/whistle-eval/samples/en2.txt @@ -0,0 +1 @@ +Please call me back at four fifteen in the afternoon. The meeting is with Doctor Smith and Professor Johnson from the university. diff --git a/tools/whistle-eval/samples/it1.txt b/tools/whistle-eval/samples/it1.txt new file mode 100644 index 0000000..680e8da --- /dev/null +++ b/tools/whistle-eval/samples/it1.txt @@ -0,0 +1 @@ +Il rapido volpe marrone salta sopra il cane pigro vicino alla riva del fiume, mentre gli scienziati misurano con attenzione la temperatura dell'apparecchiatura sperimentale. diff --git a/tools/whistle-eval/samples/it1_noisy.txt b/tools/whistle-eval/samples/it1_noisy.txt new file mode 100644 index 0000000..680e8da --- /dev/null +++ b/tools/whistle-eval/samples/it1_noisy.txt @@ -0,0 +1 @@ +Il rapido volpe marrone salta sopra il cane pigro vicino alla riva del fiume, mentre gli scienziati misurano con attenzione la temperatura dell'apparecchiatura sperimentale. diff --git a/tools/whistle-eval/samples/it2.txt b/tools/whistle-eval/samples/it2.txt new file mode 100644 index 0000000..903d626 --- /dev/null +++ b/tools/whistle-eval/samples/it2.txt @@ -0,0 +1 @@ +Per favore, richiamami alle quattro e un quarto del pomeriggio. L'incontro e con il dottor Rossi e la professoressa Bianchi dell'universita. diff --git a/tools/whistle-eval/samples/jfk.txt b/tools/whistle-eval/samples/jfk.txt new file mode 100644 index 0000000..a67a12c --- /dev/null +++ b/tools/whistle-eval/samples/jfk.txt @@ -0,0 +1 @@ +And so my fellow Americans ask not what your country can do for you ask what you can do for your country diff --git a/tools/whistle-eval/setup.sh b/tools/whistle-eval/setup.sh new file mode 100755 index 0000000..fde74fe --- /dev/null +++ b/tools/whistle-eval/setup.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Fetches everything the Whistle evaluation harness needs and rebuilds the +# ONNX external-data files from the small whistle.pack. +set -euo pipefail +cd "$(dirname "$0")" + +WHISTLE_BASE="https://huggingface.co/mrfakename/whistle-ONNX/resolve/main" +WHISPER_BASE="https://huggingface.co/csukuangfj/sherpa-onnx-whisper" + +mkdir -p vendor/js vendor/onnx models/tiny models/base + +echo "==> Downloading the Whistle pack and ONNX graphs" +curl -L --fail -o vendor/whistle.pack "$WHISTLE_BASE/whistle.pack" +curl -L --fail -o vendor/onnx/encoder.onnx "$WHISTLE_BASE/onnx/encoder.onnx" +curl -L --fail -o vendor/onnx/decoder.onnx "$WHISTLE_BASE/onnx/decoder.onnx" + +echo "==> Downloading the reference JS pipeline" +for file in pack.js whistle.js backend.js example.mjs; do + curl -L --fail -o "vendor/js/$file" "$WHISTLE_BASE/js/$file" +done + +echo "==> Installing onnxruntime-node" +npm install --no-audit --no-fund + +echo "==> Rebuilding fp32 ONNX data from whistle.pack" +node build-data.mjs + +echo "==> Downloading Whisper tiny/base int8 (sherpa-onnx)" +for model in tiny base; do + for file in encoder.int8.onnx decoder.int8.onnx tokens.txt; do + curl -L --fail -o "models/$model/$file" "$WHISPER_BASE-$model/resolve/main/$model-$file" + done +done + +echo +echo "Done. Next:" +echo " python3 -m venv .venv && .venv/bin/pip install sherpa-onnx numpy" +echo " .venv/bin/python evaluate.py \"samples/*.wav\"" diff --git a/tools/whistle-eval/whisper_baseline.py b/tools/whistle-eval/whisper_baseline.py new file mode 100644 index 0000000..9be7014 --- /dev/null +++ b/tools/whistle-eval/whisper_baseline.py @@ -0,0 +1,65 @@ +#!/usr/bin/env python3 +"""Run the same sherpa-onnx multilingual Whisper model that the app uses. + +Usage: + python whisper_baseline.py <audio.wav> [tiny|base] [language] + +Requires a virtualenv with `sherpa-onnx` installed (see README.md). The model +files are expected under ./models/<size>/. +""" +import argparse +import os +import sys +import wave + +import numpy as np +import sherpa_onnx + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("audio") + parser.add_argument("size", nargs="?", default="base", choices=["tiny", "base"]) + parser.add_argument("language", nargs="?", default="en") + args = parser.parse_args() + + model_dir = os.path.join(os.path.dirname(__file__), "models", args.size) + encoder = os.path.join(model_dir, "encoder.int8.onnx") + decoder = os.path.join(model_dir, "decoder.int8.onnx") + tokens = os.path.join(model_dir, "tokens.txt") + for path in (encoder, decoder, tokens): + if not os.path.isfile(path): + print(f"missing {path}", file=sys.stderr) + return 2 + + recognizer = sherpa_onnx.OfflineRecognizer.from_whisper( + encoder=encoder, + decoder=decoder, + tokens=tokens, + language=args.language, + task="transcribe", + num_threads=2, + decoding_method="greedy_search", + debug=False, + ) + with wave.open(args.audio, "rb") as wav: + sample_rate = wav.getframerate() + channels = wav.getnchannels() + width = wav.getsampwidth() + frames = wav.readframes(wav.getnframes()) + if width != 2: + print("only 16-bit PCM WAV is supported", file=sys.stderr) + return 2 + samples = np.frombuffer(frames, dtype=np.int16).astype(np.float32) / 32768.0 + if channels > 1: + samples = samples.reshape(-1, channels).mean(axis=1) + + stream = recognizer.create_stream() + stream.accept_waveform(sample_rate, samples) + recognizer.decode_stream(stream) + print(stream.result.text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) |
