[ { "model": "Vosk", "repo": "https://alphacephei.com/vosk/models/vosk-model-fa-0.42.zip", "family": "Vosk / Kaldi", "precision": "fp32", "params_b": 0.494933674, "gold69_wer": 43.33740831295844, "gold69_cer": 22.621107266435985, "mean_decode_ms": 754.1021553539483, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "gold69_v2_fair_wer": 41.01157830591103, "gold69_v2_fair_cer": 22.467942761568484, "gold69_v2_revision": "09f80c368937e8899c952774776d3e54cf49768c", "gold69_v2_n": 69, "visualears269_fair_wer": 37.04156479217604, "visualears269_fair_cer": 21.468009093861642, "visualears269_n": 269, "fleurs_wer": 8.61, "fleurs_cer": 3.95, "fleurs_raw_wer": 23.962719989304915, "fleurs_raw_cer": 8.96234967063516, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_fair_wer": 7.15, "visualears6669_fair_cer": 2.43, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 15.318744053282588, "fair_cer": 4.311831666091756 }, "farfield": { "count": 1934, "fair_wer": 11.879699248120302, "fair_cer": 3.0918627907269274 }, "obstructed": { "count": 4335, "fair_wer": 11.471705464108307, "fair_cer": 2.9125456709405486 } }, "visualears6669_s3": 6.23, "fleurs_s3": 5.91, "s3_avg": 6.07, "triple_threat": 5.86, "visualears6669_ess_err": 6.2, "fleurs_ess_err": 11.2, "ess_err": 8.7 }, { "model": "nvidia/stt_fa_fastconformer_hybrid_large", "repo": "https://huggingface.co/nvidia/stt_fa_fastconformer_hybrid_large", "family": "NVIDIA FastConformer Hybrid", "precision": "fp32", "params_b": 0.114621442, "gold69_wer": 73.10513447432763, "gold69_cer": 55.14705882352941, "gold69_v2_fair_wer": 53.56489945155393, "gold69_v2_fair_cer": 25.64579074521464, "gold69_v2_revision": "09f80c368937e8899c952774776d3e54cf49768c", "gold69_v2_n": 69, "mean_decode_ms": 10.000844108305836, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "visualears269_fair_wer": 44.62102689486552, "visualears269_fair_cer": 26.226047417992856, "visualears269_n": 269, "fleurs_wer": 38.69, "fleurs_cer": 25.16, "fleurs_raw_wer": 49.13914114647492, "fleurs_raw_cer": 27.509518341693358, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_fair_wer": 30.42, "visualears6669_fair_cer": 15.87, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 33.39676498572788, "fair_cer": 13.383925491548808 }, "farfield": { "count": 1934, "fair_wer": 37.13345864661654, "fair_cer": 15.907118328532436 }, "obstructed": { "count": 4335, "fair_wer": 37.58254602123308, "fair_cer": 16.342270891694255 } }, "visualears6669_s3": 29.85, "fleurs_s3": 38.17, "s3_avg": 34.01, "triple_threat": 31.42, "visualears6669_ess_err": 39.6, "fleurs_ess_err": 57.0, "ess_err": 48.3 }, { "model": "nezamisafa/whisper-persian-v4", "repo": "https://huggingface.co/nezamisafa/whisper-persian-v4", "family": "Whisper Large v3", "precision": "fp32", "params_b": 1.54349056, "gold69_wer": 34.84107579462103, "gold69_cer": 14.749134948096888, "mean_decode_ms": 390.5451996320369, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "gold69_v2_fair_wer": 31.68799512492383, "gold69_v2_fair_cer": 12.562720683887754, "gold69_v2_revision": "09f80c368937e8899c952774776d3e54cf49768c", "gold69_v2_n": 69, "visualears269_fair_wer": 45.53789731051345, "visualears269_fair_cer": 18.49626502111075, "visualears269_n": 269, "fleurs_wer": 9.71, "fleurs_cer": 4.06, "fleurs_raw_wer": 19.160435824715197, "fleurs_raw_cer": 7.194053302713482, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_fair_wer": 14.21, "visualears6669_fair_cer": 4.62, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 19.346653980336185, "fair_cer": 6.148671955846844 }, "farfield": { "count": 1934, "fair_wer": 17.5, "fair_cer": 5.271764424709745 }, "obstructed": { "count": 4335, "fair_wer": 16.902697964351805, "fair_cer": 4.987812420416924 } }, "visualears6669_s3": 11.89, "fleurs_s3": 7.36, "s3_avg": 9.62, "triple_threat": 9.04, "visualears6669_ess_err": 15.8, "fleurs_ess_err": 16.7, "ess_err": 16.25 }, { "model": "vhdm/whisper-large-fa-v1", "repo": "https://huggingface.co/vhdm/whisper-large-fa-v1", "family": "Whisper Large v3 Turbo", "precision": "fp32", "params_b": 0.80887808, "gold69_wer": 45.72127139364303, "gold69_cer": 24.134948096885815, "mean_decode_ms": 131.8458286356329, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "gold69_v2_fair_wer": 43.69287020109689, "gold69_v2_fair_cer": 21.70600260174689, "gold69_v2_revision": "09f80c368937e8899c952774776d3e54cf49768c", "gold69_v2_n": 69, "visualears269_fair_wer": 51.89486552567237, "visualears269_fair_cer": 26.161091263397207, "visualears269_n": 269, "fleurs_wer": 11.3, "fleurs_cer": 4.78, "fleurs_raw_wer": 20.388460767181368, "fleurs_raw_cer": 7.660844866138877, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_fair_wer": 14.26, "visualears6669_fair_cer": 4.5, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 19.854107199492546, "fair_cer": 6.191790272507761 }, "farfield": { "count": 1934, "fair_wer": 17.401315789473685, "fair_cer": 4.976163222178895 }, "obstructed": { "count": 4335, "fair_wer": 17.103340800623357, "fair_cer": 4.870354297356153 } }, "visualears6669_s3": 11.94, "fleurs_s3": 8.55, "s3_avg": 10.25, "triple_threat": 9.63, "visualears6669_ess_err": 17.7, "fleurs_ess_err": 18.9, "ess_err": 18.3 }, { "model": "Qwen/Qwen3-ASR-0.6B", "repo": "https://huggingface.co/Qwen/Qwen3-ASR-0.6B", "family": "Qwen3 ASR", "precision": "bf16", "params_b": 0.782426112, "gold69_wer": 76.7114914425428, "gold69_cer": 36.60611303344867, "mean_decode_ms": 294.2275273100978, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "gold69_v2_fair_wer": 76.23400365630712, "gold69_v2_fair_cer": 36.35012079539119, "gold69_v2_revision": "09f80c368937e8899c952774776d3e54cf49768c", "gold69_v2_n": 69, "visualears269_fair_wer": 77.8117359413203, "visualears269_fair_cer": 38.2104579408899, "visualears269_n": 269, "fleurs_wer": 45.76, "fleurs_cer": 20.66, "fleurs_raw_wer": 51.45481804031665, "fleurs_raw_cer": 22.794464253338976, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_fair_wer": 58.65, "visualears6669_fair_cer": 25.99, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 66.2860767522994, "fair_cer": 28.91514315281131 }, "farfield": { "count": 1934, "fair_wer": 59.934210526315795, "fair_cer": 25.368872564434774 }, "obstructed": { "count": 4335, "fair_wer": 61.01100613616441, "fair_cer": 26.52370731098857 } }, "visualears6669_s3": 55.68, "fleurs_s3": 43.1, "s3_avg": 49.39, "triple_threat": 44.74, "visualears6669_ess_err": 66.7, "fleurs_ess_err": 60.0, "ess_err": 63.35 }, { "model": "gemini-3.5-flash", "repo": "https://openrouter.ai/google/gemini-3.5-flash", "family": "Closed paid API / LLM audio", "precision": "closed", "closed_model": true, "visualears269_fair_wer": 30.684596577017114, "visualears269_fair_cer": 15.294690696541647, "visualears269_n": 269, "mean_decode_ms": 1916.4683525307387, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Closed paid OpenRouter benchmark via chat/completions audio input. Golden6669: 6,669 rows / 10.49h, reported OpenRouter cost $4.52623842. FLEURS-fa full: 4,341 rows from Reza2kn/fleurs-fa-benchmark, text column raw_transcription, reported OpenRouter cost $29.91570070. | Params encoded as 0.0 because this is a closed OpenRouter API with undisclosed parameter count.", "visualears6669_fair_wer": 3.68, "visualears6669_fair_cer": 1.26, "visualears6669_macro_wer": 8.194093522166838, "visualears6669_macro_cer": 1.6541626578679334, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 8.69013637805265, "fair_cer": 1.9920662297343912 }, "farfield": { "count": 1934, "fair_wer": 7.213345864661654, "fair_cer": 1.3836652033358912 }, "obstructed": { "count": 4335, "fair_wer": 6.060192850881465, "fair_cer": 1.2187579583075636 } }, "fleurs_wer": 4.45, "fleurs_cer": 2.35, "fleurs_raw_wer": 14.25406556469094, "fleurs_raw_cer": 4.76050039282045, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "fleurs_reported_cost": 29.915700705, "fleurs_mean_latency_s": 5.355497112144553, "visualears6669_s3": 3.1, "fleurs_s3": 3.02, "s3_avg": 3.06, "triple_threat": 3.01, "visualears6669_ess_err": 3.3, "fleurs_ess_err": 5.4, "ess_err": 4.35, "params_b": 0.0 }, { "model": "google/chirp-3", "repo": "https://openrouter.ai/google/chirp-3", "family": "Closed paid API / STT", "precision": "closed", "closed_model": true, "visualears269_fair_wer": 46.76039119804401, "visualears269_fair_cer": 36.38577691183634, "visualears269_n": 269, "mean_decode_ms": 1786.4, "status": "complete", "notes": "Closed paid OpenRouter benchmark on VisualEars269 via audio/transcriptions. Includes one upstream timeout counted as blank. Reported OpenRouter cost: $0.27040000. No FLEURS score. | Full Triple Threat scored from OpenRouter audio/transcriptions run on VisualEars6669 + FLEURS. Retried failed calls after top-up; final strict accounting counts only four repeated provider 502/timeouts as blank transcripts (VisualEars6669 3 blanks: vg6669_02270, vg6669_03184, vg6669_05997; FLEURS 1 blank: 4250). Reported OpenRouter cost from scored full-run artifacts: $28.118667. | Params encoded as 0.0 because this is a closed OpenRouter API with undisclosed parameter count.", "visualears6669_s3": 7.7, "fleurs_s3": 5.84, "s3_avg": 6.77, "triple_threat": 6.62, "ess_err": 8.88, "visualears6669_fair_wer": 8.31, "visualears6669_fair_cer": 8.77, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 6.4, "fleurs_cer": 2.11, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_ess_err": 8.82, "fleurs_ess_err": 8.95, "params_b": 0.0, "reported_cost": 28.118667 }, { "model": "microsoft/mai-transcribe-1.5", "repo": "https://openrouter.ai/microsoft/mai-transcribe-1.5", "family": "Closed paid API / STT", "precision": "closed", "closed_model": true, "visualears269_fair_wer": 113.75305623471883, "visualears269_fair_cer": 122.4062347783731, "visualears269_n": 269, "mean_decode_ms": 2010.3, "status": "complete", "notes": "Closed paid OpenRouter benchmark on VisualEars269 via audio/transcriptions. Reported OpenRouter cost: $0.10160000. No FLEURS score. | Full Triple Threat scored from OpenRouter audio/transcriptions run on VisualEars6669 + FLEURS. Retried failed calls after top-up and completed both full sets with zero provider-error blanks. Model output quality remains poor on this Persian benchmark despite complete coverage. Reported OpenRouter cost from scored full-run artifacts: $10.546900. | Params encoded as 0.0 because this is a closed OpenRouter API with undisclosed parameter count.", "visualears6669_s3": 99.99, "fleurs_s3": 99.95, "s3_avg": 99.97, "triple_threat": 109.02, "ess_err": 99.92, "visualears6669_fair_wer": 115.78, "visualears6669_fair_cer": 128.8, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 115.43, "fleurs_cer": 130.32, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_ess_err": 99.98, "fleurs_ess_err": 99.86, "params_b": 0.0, "reported_cost": 10.5469 }, { "model": "openai/gpt-audio-mini", "repo": "https://openrouter.ai/openai/gpt-audio-mini", "family": "Closed paid API / LLM audio", "precision": "closed", "closed_model": true, "visualears269_fair_wer": 202.0171149144254, "visualears269_fair_cer": 176.76570871894788, "visualears269_n": 269, "mean_decode_ms": 2385.7, "status": "complete", "notes": "Closed paid OpenRouter benchmark on VisualEars269 via chat/completions audio input. Two benchmark rows contain effectively empty one-frame audio with long references; provider rejected those invalid WAVs and they are counted as blank in the strict 269 score. Reported OpenRouter cost: $0.02792038. No FLEURS score. | Full Triple Threat scored from OpenRouter chat/completions audio-input run on VisualEars6669 + FLEURS. Completed cleanly with zero provider-error blanks on both full sets. Reported OpenRouter cost: $1.681305. | Params encoded as 0.0 because this is a closed OpenRouter API with undisclosed parameter count.", "visualears6669_s3": 21.98, "fleurs_s3": 12.58, "s3_avg": 17.28, "triple_threat": 17.91, "ess_err": 22.97, "visualears6669_fair_wer": 34.63, "visualears6669_fair_cer": 22.44, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 12.99, "fleurs_cer": 5.34, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_ess_err": 26.92, "fleurs_ess_err": 19.01, "params_b": 0.0, "reported_cost": 1.681305 }, { "model": "Chrome Web Speech API (fa-IR)", "repo": "https://developer.mozilla.org/en-US/docs/Web/API/Web_Speech_API", "family": "Closed browser API / Google Web Speech", "closed_model": true, "params_b": 0.0, "visualears6669_fair_wer": 4.17, "visualears6669_fair_cer": 1.78, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "visualears6669_by_condition": { "clean": { "count": 400, "fair_wer": 11.893434823977165, "fair_cer": 4.777509486029666 }, "farfield": { "count": 1934, "fair_wer": 12.485902255639099, "fair_cer": 6.246619454332759 }, "obstructed": { "count": 4335, "fair_wer": 10.417843576507256, "fair_cer": 4.660384909229818 } }, "fleurs_wer": 7.19, "fleurs_cer": 4.75, "mean_decode_ms": 1000.0, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Google Chrome Web Speech API capture at locale fa-IR using BlackHole virtual audio routing. Full Golden6669 run: original ws6669_* capture overlaid with wsretry78_* retry rows; 78 retry rows, 41 blank-to-text recoveries, 37 genuine blank results remain. Scored with the official Persian normalizer on all 6,669 clips. Online-only browser/cloud recognition; no local weights. Decode: 1×RT (real-time streaming) — finalizes during playback; measured per-clip wall ~7.3s/6669, ~13.6s/FLEURS. | Params encoded as 0.0 because this is a closed browser/cloud API; decode encoded as 1000 ms/sec to avoid nulls while preserving the existing 1×RT provenance.", "visualears6669_s3": 3.91, "fleurs_s3": 6.1, "s3_avg": 5.0, "triple_threat": 4.79, "ess_err": 7.4, "visualears6669_ess_err": 4.0, "fleurs_ess_err": 10.8, "precision": "closed" }, { "model": "Shenava Koochik v1.0 (114M)", "repo": "https://huggingface.co/Reza2kn/Shenava-Koochik-v1.0", "family": "Shenava (FastConformer streaming)", "precision": "fp32", "params_b": 0.114821, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "visualears6669_fair_wer": 5.23, "visualears6669_fair_cer": 1.85, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 7.12, "fleurs_cer": 3.18, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_s3": 4.73, "fleurs_s3": 5.39, "s3_avg": 5.06, "triple_threat": 4.77, "visualears6669_ess_err": 8.1, "fleurs_ess_err": 13.0, "ess_err": 10.55, "mean_decode_ms": 10.73 }, { "model": "Shenava Rizeh v1.0 (32M)", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-v1.0", "family": "Shenava (FastConformer streaming)", "precision": "fp32", "params_b": 0.032, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "visualears6669_fair_wer": 9.37, "visualears6669_fair_cer": 3.47, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 10.69, "fleurs_cer": 4.49, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_s3": 8.51, "fleurs_s3": 8.59, "s3_avg": 8.55, "triple_threat": 7.93, "ess_err": 18.1, "visualears6669_ess_err": 15.6, "fleurs_ess_err": 20.6, "mean_decode_ms": 5.75 }, { "model": "Shenava Rizeh Pizeh v1.0 (6.9M)", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-Pizeh-v1.0", "family": "Shenava (FastConformer streaming)", "precision": "fp32", "params_b": 0.0069, "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium).", "visualears6669_fair_wer": 21.53, "visualears6669_fair_cer": 8.41, "visualears6669_n": 6669, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_wer": 22.85, "fleurs_cer": 9.67, "fleurs_n": 4341, "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "visualears6669_s3": 19.69, "fleurs_s3": 19.78, "s3_avg": 19.73, "triple_threat": 18.09, "ess_err": 38.25, "visualears6669_ess_err": 34.2, "fleurs_ess_err": 42.3, "mean_decode_ms": 4.23 }, { "model": "Koochik 👯‍♂️ Vosk-small", "repo": "https://huggingface.co/Reza2kn/Shenava-Koochik-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.115, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 small (97 MB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (400ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 5.69, "visualears6669_fair_cer": 2.01, "fleurs_wer": 7.62, "fleurs_cer": 3.67, "visualears6669_s3": 4.81, "fleurs_s3": 5.04, "s3_avg": 4.92, "visualears6669_ess_err": 6.4, "fleurs_ess_err": 11.1, "ess_err": 8.75, "triple_threat": 4.85, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 412.8 }, { "model": "Koochik 👯‍♂️ Vosk-big", "repo": "https://huggingface.co/Reza2kn/Shenava-Koochik-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.115, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 big (2.8 GB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (754ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 5.03, "visualears6669_fair_cer": 1.81, "fleurs_wer": 6.98, "fleurs_cer": 3.5, "visualears6669_s3": 4.28, "fleurs_s3": 4.57, "s3_avg": 4.43, "visualears6669_ess_err": 5.3, "fleurs_ess_err": 9.8, "ess_err": 7.55, "triple_threat": 4.39, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 766.9 }, { "model": "Rizeh 👯‍♂️ Vosk-small", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.032, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 small (97 MB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (400ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 10.84, "visualears6669_fair_cer": 5.4, "fleurs_wer": 12.61, "fleurs_cer": 7.1, "visualears6669_s3": 7.21, "fleurs_s3": 6.75, "s3_avg": 6.98, "visualears6669_ess_err": 10.0, "fleurs_ess_err": 14.6, "ess_err": 12.3, "triple_threat": 7.78, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 407.9 }, { "model": "Rizeh 👯‍♂️ Vosk-big", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.032, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 big (2.8 GB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (754ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 9.69, "visualears6669_fair_cer": 5.0, "fleurs_wer": 11.54, "fleurs_cer": 6.71, "visualears6669_s3": 6.32, "fleurs_s3": 6.02, "s3_avg": 6.17, "visualears6669_ess_err": 8.1, "fleurs_ess_err": 12.6, "ess_err": 10.35, "triple_threat": 7.0, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 762.0 }, { "model": "Pizeh 👯‍♂️ Vosk-small", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-Pizeh-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.0069, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 small (97 MB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (400ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 18.71, "visualears6669_fair_cer": 9.77, "fleurs_wer": 20.78, "fleurs_cer": 11.94, "visualears6669_s3": 12.86, "fleurs_s3": 11.97, "s3_avg": 12.41, "visualears6669_ess_err": 16.9, "fleurs_ess_err": 23.5, "ess_err": 20.2, "triple_threat": 13.57, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 406.3 }, { "model": "Pizeh 👯‍♂️ Vosk-big", "repo": "https://huggingface.co/Reza2kn/Shenava-Rizeh-Pizeh-v1.0", "family": "Shenava 👯‍♂️ Vosk ensemble", "params_b": 0.0069, "precision": "fp32", "status": "complete", "notes": "COMPOUND-FAIR re-score: WER & S³ & Ess under +both normalization (ZWNJ/compound-insensitive + homophone folding), applied uniformly to every model; CER is fair_text. Full g6669 (6669) + FLEURS (4341). Gemini & Chrome per-clip freshly regenerated (Gemini via OpenRouter reasoning=medium). | Shenava base + Vosk-fa-0.42 big (2.8 GB) per-clip hotword guide (pyctcdecode beam80, hw10). S³/Ess computed with the canonical score_s3.py on full g6669 (6669) + FLEURS (4341); all-CPU, offline. (Existing rows retain their original scoring.) | decode ≈ base-fwd + 2.1ms beam + Vosk (754ms CPU); Vosk pass dominates.", "visualears6669_fair_wer": 17.0, "visualears6669_fair_cer": 9.11, "fleurs_wer": 19.23, "fleurs_cer": 11.34, "visualears6669_s3": 11.53, "fleurs_s3": 10.88, "s3_avg": 11.21, "visualears6669_ess_err": 14.5, "fleurs_ess_err": 20.9, "ess_err": 17.7, "triple_threat": 12.39, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 760.4 }, { "model": "Koochik 🧩 Static-3669", "repo": "https://huggingface.co/Reza2kn/Shenava-Koochik-v1.0", "family": "Shenava 🧩 static-hotword (offline)", "params_b": 0.115, "precision": "fp32", "status": "complete", "notes": "Fully-offline keyword recovery: Shenava Koochik base + a STATIC 3,669-word Persian hotword list shallow-fused into the CTC prefix-beam (pyctcdecode beam80, hotword_weight 3.0) — NO Vosk, no second recognizer, ~50 KB word list. S³/Ess via canonical score_norm2.py, WER via wer_corrected.py, CER via score_s3.py — compound-fair (+both), the SAME scoring applied to every row. Full g6669 (6669) + FLEURS (4341). decode ≈ base-fwd 10.7ms + 2.1ms beam (no Vosk). Beats Vosk-small on BOTH the keyword band (8.10 vs 8.75) and overall triple (4.06 vs 4.81) at ~1/2000th the decode latency and zero model footprint — best OFFLINE triple_threat on the board. Same static-list config shipped on macOS / shenava.app webapp / iOS / Android-TV / VPS.", "visualears6669_fair_wer": 4.71, "visualears6669_fair_cer": 1.67, "fleurs_wer": 6.22, "fleurs_cer": 2.94, "visualears6669_s3": 4.24, "fleurs_s3": 4.58, "s3_avg": 4.41, "visualears6669_ess_err": 6.3, "fleurs_ess_err": 9.9, "ess_err": 8.1, "triple_threat": 4.2, "visualears6669_n": 6669, "fleurs_n": 4341, "visualears6669_dataset": "Reza2kn/visualears-golden-6669", "fleurs_dataset": "Reza2kn/fleurs-fa-benchmark", "fleurs_text_column": "raw_transcription", "mean_decode_ms": 12.8 } ]