NaijaVox-2.0 — merged standalone model (epoch 3, r=64, SpecAugment)
Browse files- README.md +374 -0
- config.json +47 -0
- generation_config.json +9 -0
- model.safetensors +3 -0
- processor_config.json +17 -0
- tokenizer.json +0 -0
- tokenizer_config.json +126 -0
README.md
ADDED
|
@@ -0,0 +1,374 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
language:
|
| 3 |
+
- yo
|
| 4 |
+
- ha
|
| 5 |
+
- ig
|
| 6 |
+
- en
|
| 7 |
+
- pcm
|
| 8 |
+
license: apache-2.0
|
| 9 |
+
tags:
|
| 10 |
+
- whisper
|
| 11 |
+
- speech
|
| 12 |
+
- asr
|
| 13 |
+
- automatic-speech-recognition
|
| 14 |
+
- yoruba
|
| 15 |
+
- hausa
|
| 16 |
+
- igbo
|
| 17 |
+
- nigerian-english
|
| 18 |
+
- nigerian-pidgin
|
| 19 |
+
- nigeria
|
| 20 |
+
- african-languages
|
| 21 |
+
- audio
|
| 22 |
+
base_model: openai/whisper-large-v3
|
| 23 |
+
datasets:
|
| 24 |
+
- google/fleurs
|
| 25 |
+
- benjaminogbonna/nigerian_accented_english_dataset
|
| 26 |
+
- asr-nigerian-pidgin/nigerian-pidgin-1.0
|
| 27 |
+
- Tundragoon/IroyinSpeech
|
| 28 |
+
- google/WaxalNLP
|
| 29 |
+
- benjaminogbonna/nigerian_common_voice_dataset
|
| 30 |
+
- vpetukhov/bible_tts_hausa
|
| 31 |
+
pipeline_tag: automatic-speech-recognition
|
| 32 |
+
---
|
| 33 |
+
|
| 34 |
+
<p align="center">
|
| 35 |
+
<img src="https://huggingface.co/Axiveri/NaijaVox-V1/resolve/main/naijavox_logo.png" width="100%"/>
|
| 36 |
+
</p>
|
| 37 |
+
|
| 38 |
+
<p align="center">
|
| 39 |
+
<img src="https://img.shields.io/badge/Model-NaijaVox--2.0-2ea44f?style=flat-square"/>
|
| 40 |
+
<img src="https://img.shields.io/badge/License-Apache%202.0-2196F3?style=flat-square"/>
|
| 41 |
+
<img src="https://img.shields.io/badge/Base-Whisper--Large--v3-8B5CF6?style=flat-square"/>
|
| 42 |
+
<img src="https://img.shields.io/badge/Version-2.0-FF9800?style=flat-square"/>
|
| 43 |
+
</p>
|
| 44 |
+
|
| 45 |
+
<p align="center">
|
| 46 |
+
<img src="https://img.shields.io/badge/Languages-Yoruba%20%7C%20Hausa%20%7C%20Igbo%20%7C%20Pidgin%20%7C%20Naija%20English-FF6B35?style=flat-square"/>
|
| 47 |
+
</p>
|
| 48 |
+
|
| 49 |
+
<p align="center">
|
| 50 |
+
<img src="https://img.shields.io/badge/Built%20by-Axiveri-e11d48?style=flat-square"/>
|
| 51 |
+
<img src="https://img.shields.io/badge/GPU-Tesla%20T4%20x2-64748b?style=flat-square"/>
|
| 52 |
+
</p>
|
| 53 |
+
|
| 54 |
+
<div align="center">
|
| 55 |
+
|
| 56 |
+
| Language | WER | vs V1 |
|
| 57 |
+
|---|---|---|
|
| 58 |
+
| 🇳🇬 Pidgin | **14.7%** | ↓ 2.1pp |
|
| 59 |
+
| 🇳🇬 Nigerian English | **19.6%** | ↓ 1.5pp |
|
| 60 |
+
| 🇳🇬 Yoruba | **22.3%** | ↓ 6.5pp |
|
| 61 |
+
| 🇳🇬 Hausa | **25.8%** | ↓ 5.2pp |
|
| 62 |
+
| 🇳🇬 Igbo | **30.5%** | ↓ 11.4pp |
|
| 63 |
+
|
| 64 |
+
</div>
|
| 65 |
+
|
| 66 |
+
---
|
| 67 |
+
|
| 68 |
+
## Nigeria's Voice in AI. Now Sharper.
|
| 69 |
+
|
| 70 |
+
**NaijaVox-2.0** is the second generation of Axiveri's open-weight automatic speech recognition model for Nigerian languages — Yoruba (with full diacritics), Hausa, Igbo, Nigerian Pidgin, and Nigerian-accented English. Built on OpenAI Whisper-large-v3 with PEFT LoRA fine-tuning, NaijaVox-2.0 delivers significant accuracy gains over V1 through a larger and more diverse training corpus (25,866 samples across 7 datasets), deeper LoRA adaptation (r=64 targeting attention and feed-forward layers), SpecAugment, and realistic noise augmentation for real-world robustness.
|
| 71 |
+
|
| 72 |
+
> *"Every Nigerian deserves to be heard and understood by AI — in their own language, with their own voice."*
|
| 73 |
+
|
| 74 |
+
**[← NaijaVox-V1](https://huggingface.co/Axiveri/NaijaVox-V1)** — the original model
|
| 75 |
+
|
| 76 |
+
---
|
| 77 |
+
|
| 78 |
+
## 📈 V1 → V2 Improvement
|
| 79 |
+
|
| 80 |
+
Evaluated on identical test sets with identical methodology (50 samples/language, strict WER, no normalization):
|
| 81 |
+
|
| 82 |
+
| Language | V1 WER | V2 WER | Absolute Δ | Relative Gain |
|
| 83 |
+
|---|---|---|---|---|
|
| 84 |
+
| 🇳🇬 Yoruba | 28.8% | **22.3%** | −6.5pp | **+22.6%** |
|
| 85 |
+
| 🇳🇬 Hausa | 31.0% | **25.8%** | −5.2pp | **+16.8%** |
|
| 86 |
+
| 🇳🇬 Igbo | 41.9% | **30.5%** | −11.4pp | **+27.2%** |
|
| 87 |
+
| 🇳🇬 Nigerian English | 21.1% | **19.6%** | −1.5pp | **+7.1%** |
|
| 88 |
+
| 🇳🇬 Nigerian Pidgin | 16.8% | **14.7%** | −2.1pp | **+12.5%** |
|
| 89 |
+
| **Average** | **27.9%** | **22.58%** | **−5.3pp** | **+19.1%** |
|
| 90 |
+
|
| 91 |
+
> Igbo sees the largest jump (+27.2% relative) — driven by WaxalNLP Igbo TTS data and Nigerian Common Voice Igbo samples, combined with SpecAugment frequency masking.
|
| 92 |
+
|
| 93 |
+
---
|
| 94 |
+
|
| 95 |
+
## 🗣️ Languages Supported
|
| 96 |
+
|
| 97 |
+
| Language | ISO Code | Script | Token |
|
| 98 |
+
|---|---|---|---|
|
| 99 |
+
| Yoruba | `yo` | Latin + full diacritics (ẹ, ọ, ṣ, à, á, etc.) | `<|yo|>` |
|
| 100 |
+
| Hausa | `ha` | Latin + special chars (ƙ, ƴ, ɗ, etc.) | `<|ha|>` |
|
| 101 |
+
| Igbo | `ig` | Latin + diacritics | `<|ig|>` |
|
| 102 |
+
| Nigerian Pidgin | `pcm` | Latin | `<|pcm|>` |
|
| 103 |
+
| Nigerian English | `en` | Latin | `<|en|>` |
|
| 104 |
+
|
| 105 |
+
> **Note:** `<|ig|>` and `<|pcm|>` are custom language tokens added to the Whisper vocabulary. The extended tokenizer is included in this repository.
|
| 106 |
+
|
| 107 |
+
---
|
| 108 |
+
|
| 109 |
+
## 🚀 Quick Start
|
| 110 |
+
|
| 111 |
+
```python
|
| 112 |
+
from transformers import pipeline
|
| 113 |
+
|
| 114 |
+
pipe = pipeline(
|
| 115 |
+
"automatic-speech-recognition",
|
| 116 |
+
model="Axiveri/NaijaVox-2.0",
|
| 117 |
+
device=0 # use GPU, or remove for CPU
|
| 118 |
+
)
|
| 119 |
+
|
| 120 |
+
result = pipe("your_audio.wav")
|
| 121 |
+
print(result["text"])
|
| 122 |
+
```
|
| 123 |
+
|
| 124 |
+
### Specifying Language
|
| 125 |
+
|
| 126 |
+
```python
|
| 127 |
+
from transformers import WhisperProcessor, WhisperForConditionalGeneration
|
| 128 |
+
import torch
|
| 129 |
+
|
| 130 |
+
model = WhisperForConditionalGeneration.from_pretrained("Axiveri/NaijaVox-2.0")
|
| 131 |
+
processor = WhisperProcessor.from_pretrained("Axiveri/NaijaVox-2.0")
|
| 132 |
+
vocab = processor.tokenizer.get_vocab()
|
| 133 |
+
|
| 134 |
+
LANG_TOKENS = {
|
| 135 |
+
"yoruba": "<|yo|>",
|
| 136 |
+
"hausa": "<|ha|>",
|
| 137 |
+
"igbo": "<|ig|>",
|
| 138 |
+
"nigerian_english": "<|en|>",
|
| 139 |
+
"pidgin": "<|pcm|>",
|
| 140 |
+
}
|
| 141 |
+
|
| 142 |
+
def transcribe(audio_array, sampling_rate, language="yoruba"):
|
| 143 |
+
lang_id = vocab[LANG_TOKENS[language]]
|
| 144 |
+
transcribe = vocab["<|transcribe|>"]
|
| 145 |
+
notimestamps = vocab["<|notimestamps|>"]
|
| 146 |
+
forced_ids = [[1, lang_id], [2, transcribe], [3, notimestamps]]
|
| 147 |
+
|
| 148 |
+
inputs = processor.feature_extractor(
|
| 149 |
+
audio_array, sampling_rate=sampling_rate, return_tensors="pt"
|
| 150 |
+
).input_features
|
| 151 |
+
|
| 152 |
+
with torch.no_grad():
|
| 153 |
+
generated = model.generate(
|
| 154 |
+
input_features=inputs,
|
| 155 |
+
forced_decoder_ids=forced_ids,
|
| 156 |
+
max_new_tokens=448
|
| 157 |
+
)
|
| 158 |
+
return processor.tokenizer.decode(generated[0], skip_special_tokens=True)
|
| 159 |
+
```
|
| 160 |
+
|
| 161 |
+
---
|
| 162 |
+
|
| 163 |
+
## 📊 Benchmark Results
|
| 164 |
+
|
| 165 |
+
Evaluated on FLEURS test splits (Yoruba, Hausa, Igbo), Nigerian Pidgin ASR test set, and Nigerian Accented English dataset. 50 samples per language, greedy decoding, strict WER via `jiwer` (no text normalization). **Identical methodology to V1 for direct comparison.**
|
| 166 |
+
|
| 167 |
+
| Language | WER (%) | Accuracy (%) | Test Set | Samples |
|
| 168 |
+
|---|---|---|---|---|
|
| 169 |
+
| 🇳🇬 Nigerian Pidgin | **14.7** | **85.3** | asr-nigerian-pidgin/nigerian-pidgin-1.0 | 50 |
|
| 170 |
+
| 🇳🇬 Nigerian English | **19.6** | **80.4** | benjaminogbonna/nigerian_accented_english | 50 |
|
| 171 |
+
| 🇳🇬 Yoruba | **22.3** | **77.7** | google/fleurs yo_ng | 50 |
|
| 172 |
+
| 🇳🇬 Hausa | **25.8** | **74.2** | google/fleurs ha_ng | 50 |
|
| 173 |
+
| 🇳🇬 Igbo | **30.5** | **70.5** | google/fleurs ig_ng | 50 |
|
| 174 |
+
| **Average** | **22.58** | **77.62** | — | 250 |
|
| 175 |
+
|
| 176 |
+
> Lower WER = better. Human-level transcription ≈ 5–10%.
|
| 177 |
+
|
| 178 |
+
---
|
| 179 |
+
|
| 180 |
+
## 🛡️ Robustness Improvements over V1
|
| 181 |
+
|
| 182 |
+
NaijaVox-2.0 was trained with two techniques not present in V1:
|
| 183 |
+
|
| 184 |
+
### SpecAugment
|
| 185 |
+
Frequency masking (up to 27 mel bins) and time masking (up to 100 time steps) applied to mel spectrograms during training. This prevents over-reliance on specific frequency bands or time positions, improving generalization to real-world recordings where background noise occupies variable frequency ranges.
|
| 186 |
+
|
| 187 |
+
### Noise Augmentation
|
| 188 |
+
30% of training samples received realistic background noise injection at random SNR levels before mel extraction. This directly trains the model for common Nigerian recording conditions — market noise, phone compression artifacts, outdoor ambient sound, and crowd audio — conditions where V1 degraded noticeably.
|
| 189 |
+
|
| 190 |
+
### Code-Switching Robustness
|
| 191 |
+
NaijaVox-2.0 was trained on Nigerian Pidgin and Nigerian English together with Yoruba, Hausa, and Igbo — all of which contain natural Yoruba-English, Igbo-English, and Pidgin-English code-switching patterns present in everyday Nigerian speech, media, and social content. The model handles mid-sentence language shifts without explicit code-switching supervision.
|
| 192 |
+
|
| 193 |
+
---
|
| 194 |
+
|
| 195 |
+
## 🎙️ Sample Transcriptions
|
| 196 |
+
|
| 197 |
+
> Real audio samples from FLEURS test, Nigerian English, and Pidgin datasets — data the model **never saw during training**. Transcriptions generated live by the published merged model.
|
| 198 |
+
|
| 199 |
+
### Yoruba
|
| 200 |
+
|
| 201 |
+
| Reference | Audio | NaijaVox-2.0 Output |
|
| 202 |
+
|---|:---:|---|
|
| 203 |
+
| *àwọn èyàn ti mọ̀ nípa àwọn kemika pepe bí wúrà fàdákà àti kọ́pa àtijọ́ torípé a lè rí wọn* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/yoruba_sample1.wav" type="audio/wav"></audio> | àwọn èèyàn ti mọ̀ nípa àwọn kẹmíkà pèèpèé bí wúrà fàdákà àti kọpa àtijọ́ torí pé a lè rí wọn |
|
| 204 |
+
| *àwọn ara ìrano lo kọ́kọ́ bẹ̀rẹ̀ si ni sin ewure ní bíi ọdún 15,0000 sẹ́yìn ní oke sagrosi* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/yoruba_sample2.wav" type="audio/wav"></audio> | àwọn ará ìrà náà ló kọ́kọ́ bẹ̀rẹ̀ sí ní sin ewúrẹ́ ní bí ọdún 1500 sẹ́yìn ní òkè sagrosi |
|
| 205 |
+
|
| 206 |
+
### Hausa
|
| 207 |
+
|
| 208 |
+
| Reference | Audio | NaijaVox-2.0 Output |
|
| 209 |
+
|---|:---:|---|
|
| 210 |
+
| *an kwatanta faretin gine-ginen da ke yin sararin samaniyar hong kong da ginshiƙi mai walƙi* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/hausa_sample1.wav" type="audio/wav"></audio> | an kwatanta feretin gine-ginen da ke yin sararin samaniya hong kong da ginshiki mai walƙiy |
|
| 211 |
+
| *aristotle masanin falsafa ne yayi tunanin cewa komai ya kunshi cakuda daya ko fiye daga ab* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/hausa_sample2.wav" type="audio/wav"></audio> | aristotle masanin falsafani ya yi tunanin cewa kome ya kunshi ca kuda daya ko fiye daga ab |
|
| 212 |
+
|
| 213 |
+
### Igbo
|
| 214 |
+
|
| 215 |
+
| Reference | Audio | NaijaVox-2.0 Output |
|
| 216 |
+
|---|:---:|---|
|
| 217 |
+
| *ka akara rossby na-adị obere karịa ka arụmarụ na-adịkwu obere nke kpakpando n'ikwanye ugwu* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/igbo_sample1.wav" type="audio/wav"></audio> | akara rossby na-adị obere karịa ka arụmarụ na-adịkwa obere nke kpakpando n'ịkwà nye monto |
|
| 218 |
+
| *ka agha dara mba britenị jiri ndị agha elu mmiri gbochie ndị jamani inweta enyemaka* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/igbo_sample2.wav" type="audio/wav"></audio> | ka agha adara mba briten jiri ndị agha elu mmiri gbochie ndị jamanị inweta enyemaka |
|
| 219 |
+
|
| 220 |
+
### Nigerian English
|
| 221 |
+
|
| 222 |
+
| Reference | Audio | NaijaVox-2.0 Output |
|
| 223 |
+
|---|:---:|---|
|
| 224 |
+
| *Did it change plain? Yes. yes. Ok that means he was correct so this is if he's right that* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/english_sample1.wav" type="audio/wav"></audio> | Did it change green? Yes. Ok that means she was correct. So this is if its red then its no |
|
| 225 |
+
| *Ebube Nwagbo studied Mass Communication at Nnamdi Azikiwe University.* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/english_sample2.wav" type="audio/wav"></audio> | Ebube Nwagbo studied Mass Communication at Nnamdi Azikiwe University. |
|
| 226 |
+
|
| 227 |
+
### Nigerian Pidgin
|
| 228 |
+
|
| 229 |
+
| Reference | Audio | NaijaVox-2.0 Output |
|
| 230 |
+
|---|:---:|---|
|
| 231 |
+
| *on top di injury her uncle no even carry her go hospital for treatment* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/pidgin_sample1.wav" type="audio/wav"></audio> | on top di injury and her uncle no even carry her go hospital for treatment |
|
| 232 |
+
| *she tell don jazzy for december 2016 say as she be* | <audio controls><source src="https://huggingface.co/Axiveri/NaijaVox-2.0/resolve/main/audio_samples/pidgin_sample2.wav" type="audio/wav"></audio> | she tell don jazzy for december 2016 say i should be |
|
| 233 |
+
|
| 234 |
+
---
|
| 235 |
+
|
| 236 |
+
## 🏗️ Model Architecture
|
| 237 |
+
|
| 238 |
+
```
|
| 239 |
+
Input Audio (16kHz)
|
| 240 |
+
│
|
| 241 |
+
▼
|
| 242 |
+
Whisper-large-v3 Encoder (frozen during fine-tuning)
|
| 243 |
+
│ 1500 × 1280 features
|
| 244 |
+
▼
|
| 245 |
+
Whisper Decoder + LoRA (r=64, alpha=128, fine-tuned)
|
| 246 |
+
target modules: q_proj, k_proj, v_proj, out_proj, fc1, fc2
|
| 247 |
+
V1 target: attention only (q/k/v/out)
|
| 248 |
+
V2 target: attention + feed-forward (adds fc1/fc2)
|
| 249 |
+
│
|
| 250 |
+
▼
|
| 251 |
+
Extended Tokenizer (vocab: 51,868 tokens)
|
| 252 |
+
+ <|ig|> Igbo token
|
| 253 |
+
+ <|pcm|> Nigerian Pidgin token
|
| 254 |
+
│
|
| 255 |
+
▼
|
| 256 |
+
Transcript
|
| 257 |
+
```
|
| 258 |
+
|
| 259 |
+
> V2 publishes a **fully merged standalone model** — no PEFT dependency required. Load directly with `transformers`.
|
| 260 |
+
|
| 261 |
+
---
|
| 262 |
+
|
| 263 |
+
## 📦 Training Details
|
| 264 |
+
|
| 265 |
+
| Parameter | V1 | V2 |
|
| 266 |
+
|---|---|---|
|
| 267 |
+
| Base model | openai/whisper-large-v3 | openai/whisper-large-v3 |
|
| 268 |
+
| Fine-tuning method | LoRA (PEFT) | LoRA (PEFT) |
|
| 269 |
+
| LoRA rank | 32 | **64** |
|
| 270 |
+
| LoRA alpha | 64 | **128** |
|
| 271 |
+
| Target modules | q/k/v/out_proj | **q/k/v/out_proj + fc1/fc2** |
|
| 272 |
+
| LoRA dropout | 0.05 | 0.05 |
|
| 273 |
+
| Training precision | fp16 | fp16 |
|
| 274 |
+
| Effective batch size | 16 | **32** |
|
| 275 |
+
| Learning rate | 1e-3 | **5e-4** |
|
| 276 |
+
| Warmup steps | 50 | **200** |
|
| 277 |
+
| Epochs (best) | 2 | **3 of 5** |
|
| 278 |
+
| SpecAugment | ❌ | **✅** |
|
| 279 |
+
| Noise augmentation | ❌ | **✅ (30% of samples)** |
|
| 280 |
+
| Total training samples | 13,866 | **25,866** |
|
| 281 |
+
| GPU | Tesla T4 × 2 (Kaggle) | Tesla T4 × 2 (Kaggle) |
|
| 282 |
+
| Total training time | ~20 hours | ~40 hours |
|
| 283 |
+
|
| 284 |
+
### Training Datasets
|
| 285 |
+
|
| 286 |
+
| Dataset | Language(s) | Samples | License | New in V2 |
|
| 287 |
+
|---|---|---|---|---|
|
| 288 |
+
| google/fleurs (yo_ng) | Yoruba | 2,339 | CC-BY 4.0 | — |
|
| 289 |
+
| google/fleurs (ha_ng) | Hausa | 3,259 | CC-BY 4.0 | — |
|
| 290 |
+
| google/fleurs (ig_ng) | Igbo | 2,839 | CC-BY 4.0 | — |
|
| 291 |
+
| benjaminogbonna/nigerian_accented_english_dataset | Nigerian English | 2,721 | Apache 2.0 | — |
|
| 292 |
+
| asr-nigerian-pidgin/nigerian-pidgin-1.0 | Nigerian Pidgin | 2,708 | CC-BY 4.0 | — |
|
| 293 |
+
| Tundragoon/IroyinSpeech | Yoruba | 2,500 | CC-BY 4.0 | ✅ |
|
| 294 |
+
| google/WaxalNLP (ha/ig/yo/pcm) | Hausa, Igbo, Yoruba, Pidgin | 6,000 | CC-BY-SA 4.0 | ✅ |
|
| 295 |
+
| benjaminogbonna/nigerian_common_voice_dataset | en/ha/ig/yo | 2,000 | Apache 2.0 | ✅ |
|
| 296 |
+
| vpetukhov/bible_tts_hausa | Hausa | 1,500 | CC-BY-SA | ✅ |
|
| 297 |
+
| **Total** | **5 languages** | **25,866** | | |
|
| 298 |
+
|
| 299 |
+
---
|
| 300 |
+
|
| 301 |
+
## ✅ Intended Use
|
| 302 |
+
|
| 303 |
+
NaijaVox-2.0 is designed for:
|
| 304 |
+
|
| 305 |
+
- 🏦 **Fintech & banking** — voice-based transactions and customer service in Nigerian languages
|
| 306 |
+
- 📱 **Mobile apps** — voice input for Yoruba, Hausa, Igbo, and Pidgin speakers
|
| 307 |
+
- 🎙️ **Media & journalism** — transcribing interviews and broadcasts in Nigerian languages
|
| 308 |
+
- 🏥 **Healthcare** — patient intake and medical documentation
|
| 309 |
+
- 📚 **Education** — language learning tools and accessibility
|
| 310 |
+
- 🔬 **Research** — low-resource ASR study for West African languages
|
| 311 |
+
- ♿ **Accessibility** — assistive technology for Nigerians with disabilities
|
| 312 |
+
|
| 313 |
+
---
|
| 314 |
+
|
| 315 |
+
## 🚫 Prohibited Use
|
| 316 |
+
|
| 317 |
+
The following uses are explicitly **prohibited** under this model's responsible use policy:
|
| 318 |
+
|
| 319 |
+
- ❌ **Non-consensual surveillance** — transcribing calls or conversations without consent of all parties
|
| 320 |
+
- ❌ **Fraud facilitation** — using transcription output to forge spoken statements or support advance-fee fraud
|
| 321 |
+
- ❌ **Deepfake pipelines** — combining with TTS to create fake audio-text pairs attributed to real people
|
| 322 |
+
- ❌ **Discriminatory systems** — denying services based on language or accent identification from this model
|
| 323 |
+
- ❌ **Political disinformation** — generating or verifying false transcripts of political speech
|
| 324 |
+
|
| 325 |
+
While Apache 2.0 permits broad commercial use, these restrictions apply as a binding behavioral restriction under Axiveri's Responsible AI Use Policy.
|
| 326 |
+
|
| 327 |
+
---
|
| 328 |
+
|
| 329 |
+
## 👤 Creator
|
| 330 |
+
|
| 331 |
+
**Emmanuel Ariyo (Ememzyvisuals)** — Founder, Axiveri
|
| 332 |
+
|
| 333 |
+
NaijaVox is conceived, built, and trained by Emmanuel Ariyo — combining ML engineering with a Nigerian cultural design identity to bring open-weight speech recognition to Yoruba, Hausa, Igbo, Nigerian Pidgin, and Nigerian English speakers.
|
| 334 |
+
|
| 335 |
+
---
|
| 336 |
+
|
| 337 |
+
## 👥 About Axiveri
|
| 338 |
+
|
| 339 |
+
**Axiveri** is building Africa's AI infrastructure — open models, open data, and open tools for African languages and developers.
|
| 340 |
+
|
| 341 |
+
- 🌍 [Africlaude Series](https://huggingface.co/Axiveri) — African language models
|
| 342 |
+
- 🗣️ [NaijaVox Collection](https://huggingface.co/collections/Axiveri/naijavox-nigerian-speech-recognition) — all NaijaVox versions
|
| 343 |
+
|
| 344 |
+
---
|
| 345 |
+
|
| 346 |
+
## 📄 Citation
|
| 347 |
+
|
| 348 |
+
```bibtex
|
| 349 |
+
@misc{naijavox2026,
|
| 350 |
+
title = {NaijaVox-2.0: Open-Weight Speech Recognition for Nigerian Languages},
|
| 351 |
+
author = {Ariyo, Emmanuel (Ememzyvisuals)},
|
| 352 |
+
year = {2026},
|
| 353 |
+
publisher = {HuggingFace},
|
| 354 |
+
howpublished = {\url{https://huggingface.co/Axiveri/NaijaVox-2.0}},
|
| 355 |
+
note = {Whisper-large-v3 fine-tuned on Yoruba, Hausa, Igbo,
|
| 356 |
+
Nigerian Pidgin and Nigerian-accented English.
|
| 357 |
+
LoRA r=64, SpecAugment, noise augmentation, 25,866 samples.}
|
| 358 |
+
}
|
| 359 |
+
```
|
| 360 |
+
|
| 361 |
+
---
|
| 362 |
+
|
| 363 |
+
## 📜 License
|
| 364 |
+
|
| 365 |
+
**Apache 2.0** — free for commercial and research use with attribution.
|
| 366 |
+
Additional behavioral restrictions apply as described in the Prohibited Use section above.
|
| 367 |
+
|
| 368 |
+
---
|
| 369 |
+
|
| 370 |
+
<p align="center">
|
| 371 |
+
<i>Built in Nigeria 🇳🇬 — for Nigeria and the world.</i><br/>
|
| 372 |
+
<i>Created by <a href="https://huggingface.co/ememzyvisuals">Emmanuel Ariyo (Ememzyvisuals)</a></i><br/>
|
| 373 |
+
<i>Second model in the NaijaVox series by <a href="https://huggingface.co/Axiveri">Axiveri</a></i>
|
| 374 |
+
</p>
|
config.json
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation_dropout": 0.0,
|
| 3 |
+
"activation_function": "gelu",
|
| 4 |
+
"apply_spec_augment": false,
|
| 5 |
+
"architectures": [
|
| 6 |
+
"WhisperForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"bos_token_id": 50257,
|
| 10 |
+
"classifier_proj_size": 256,
|
| 11 |
+
"d_model": 1280,
|
| 12 |
+
"decoder_attention_heads": 20,
|
| 13 |
+
"decoder_ffn_dim": 5120,
|
| 14 |
+
"decoder_layerdrop": 0.0,
|
| 15 |
+
"decoder_layers": 32,
|
| 16 |
+
"decoder_start_token_id": 50258,
|
| 17 |
+
"dropout": 0.0,
|
| 18 |
+
"dtype": "float16",
|
| 19 |
+
"encoder_attention_heads": 20,
|
| 20 |
+
"encoder_ffn_dim": 5120,
|
| 21 |
+
"encoder_layerdrop": 0.0,
|
| 22 |
+
"encoder_layers": 32,
|
| 23 |
+
"eos_token_id": 50257,
|
| 24 |
+
"forced_decoder_ids": null,
|
| 25 |
+
"init_std": 0.02,
|
| 26 |
+
"is_encoder_decoder": true,
|
| 27 |
+
"mask_feature_length": 10,
|
| 28 |
+
"mask_feature_min_masks": 0,
|
| 29 |
+
"mask_feature_prob": 0.0,
|
| 30 |
+
"mask_time_length": 10,
|
| 31 |
+
"mask_time_min_masks": 2,
|
| 32 |
+
"mask_time_prob": 0.05,
|
| 33 |
+
"max_source_positions": 1500,
|
| 34 |
+
"max_target_positions": 448,
|
| 35 |
+
"median_filter_width": 7,
|
| 36 |
+
"model_type": "whisper",
|
| 37 |
+
"num_hidden_layers": 32,
|
| 38 |
+
"num_mel_bins": 128,
|
| 39 |
+
"pad_token_id": 50256,
|
| 40 |
+
"scale_embedding": false,
|
| 41 |
+
"tie_word_embeddings": true,
|
| 42 |
+
"transformers_version": "5.0.0",
|
| 43 |
+
"use_cache": true,
|
| 44 |
+
"use_weighted_layer_sum": false,
|
| 45 |
+
"vocab_size": 51868,
|
| 46 |
+
"suppress_tokens": []
|
| 47 |
+
}
|
generation_config.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token_id": 50257,
|
| 3 |
+
"decoder_start_token_id": 50258,
|
| 4 |
+
"eos_token_id": 50257,
|
| 5 |
+
"forced_decoder_ids": null,
|
| 6 |
+
"max_new_tokens": 448,
|
| 7 |
+
"suppress_tokens": [],
|
| 8 |
+
"transformers_version": "5.0.0"
|
| 9 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7d026e7088605bf2cfa35c8e00985ab4284f6a02ef29e80c2d2465556a6309f7
|
| 3 |
+
size 3087136096
|
processor_config.json
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"feature_extractor": {
|
| 3 |
+
"chunk_length": 30,
|
| 4 |
+
"dither": 0.0,
|
| 5 |
+
"feature_extractor_type": "WhisperFeatureExtractor",
|
| 6 |
+
"feature_size": 128,
|
| 7 |
+
"hop_length": 160,
|
| 8 |
+
"n_fft": 400,
|
| 9 |
+
"n_samples": 480000,
|
| 10 |
+
"nb_max_frames": 3000,
|
| 11 |
+
"padding_side": "right",
|
| 12 |
+
"padding_value": 0.0,
|
| 13 |
+
"return_attention_mask": false,
|
| 14 |
+
"sampling_rate": 16000
|
| 15 |
+
},
|
| 16 |
+
"processor_class": "WhisperProcessor"
|
| 17 |
+
}
|
tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"clean_up_tokenization_spaces": true,
|
| 6 |
+
"eos_token": "<|endoftext|>",
|
| 7 |
+
"errors": "replace",
|
| 8 |
+
"extra_special_tokens": [
|
| 9 |
+
"<|startoftranscript|>",
|
| 10 |
+
"<|en|>",
|
| 11 |
+
"<|zh|>",
|
| 12 |
+
"<|de|>",
|
| 13 |
+
"<|es|>",
|
| 14 |
+
"<|ru|>",
|
| 15 |
+
"<|ko|>",
|
| 16 |
+
"<|fr|>",
|
| 17 |
+
"<|ja|>",
|
| 18 |
+
"<|pt|>",
|
| 19 |
+
"<|tr|>",
|
| 20 |
+
"<|pl|>",
|
| 21 |
+
"<|ca|>",
|
| 22 |
+
"<|nl|>",
|
| 23 |
+
"<|ar|>",
|
| 24 |
+
"<|sv|>",
|
| 25 |
+
"<|it|>",
|
| 26 |
+
"<|id|>",
|
| 27 |
+
"<|hi|>",
|
| 28 |
+
"<|fi|>",
|
| 29 |
+
"<|vi|>",
|
| 30 |
+
"<|he|>",
|
| 31 |
+
"<|uk|>",
|
| 32 |
+
"<|el|>",
|
| 33 |
+
"<|ms|>",
|
| 34 |
+
"<|cs|>",
|
| 35 |
+
"<|ro|>",
|
| 36 |
+
"<|da|>",
|
| 37 |
+
"<|hu|>",
|
| 38 |
+
"<|ta|>",
|
| 39 |
+
"<|no|>",
|
| 40 |
+
"<|th|>",
|
| 41 |
+
"<|ur|>",
|
| 42 |
+
"<|hr|>",
|
| 43 |
+
"<|bg|>",
|
| 44 |
+
"<|lt|>",
|
| 45 |
+
"<|la|>",
|
| 46 |
+
"<|mi|>",
|
| 47 |
+
"<|ml|>",
|
| 48 |
+
"<|cy|>",
|
| 49 |
+
"<|sk|>",
|
| 50 |
+
"<|te|>",
|
| 51 |
+
"<|fa|>",
|
| 52 |
+
"<|lv|>",
|
| 53 |
+
"<|bn|>",
|
| 54 |
+
"<|sr|>",
|
| 55 |
+
"<|az|>",
|
| 56 |
+
"<|sl|>",
|
| 57 |
+
"<|kn|>",
|
| 58 |
+
"<|et|>",
|
| 59 |
+
"<|mk|>",
|
| 60 |
+
"<|br|>",
|
| 61 |
+
"<|eu|>",
|
| 62 |
+
"<|is|>",
|
| 63 |
+
"<|hy|>",
|
| 64 |
+
"<|ne|>",
|
| 65 |
+
"<|mn|>",
|
| 66 |
+
"<|bs|>",
|
| 67 |
+
"<|kk|>",
|
| 68 |
+
"<|sq|>",
|
| 69 |
+
"<|sw|>",
|
| 70 |
+
"<|gl|>",
|
| 71 |
+
"<|mr|>",
|
| 72 |
+
"<|pa|>",
|
| 73 |
+
"<|si|>",
|
| 74 |
+
"<|km|>",
|
| 75 |
+
"<|sn|>",
|
| 76 |
+
"<|yo|>",
|
| 77 |
+
"<|so|>",
|
| 78 |
+
"<|af|>",
|
| 79 |
+
"<|oc|>",
|
| 80 |
+
"<|ka|>",
|
| 81 |
+
"<|be|>",
|
| 82 |
+
"<|tg|>",
|
| 83 |
+
"<|sd|>",
|
| 84 |
+
"<|gu|>",
|
| 85 |
+
"<|am|>",
|
| 86 |
+
"<|yi|>",
|
| 87 |
+
"<|lo|>",
|
| 88 |
+
"<|uz|>",
|
| 89 |
+
"<|fo|>",
|
| 90 |
+
"<|ht|>",
|
| 91 |
+
"<|ps|>",
|
| 92 |
+
"<|tk|>",
|
| 93 |
+
"<|nn|>",
|
| 94 |
+
"<|mt|>",
|
| 95 |
+
"<|sa|>",
|
| 96 |
+
"<|lb|>",
|
| 97 |
+
"<|my|>",
|
| 98 |
+
"<|bo|>",
|
| 99 |
+
"<|tl|>",
|
| 100 |
+
"<|mg|>",
|
| 101 |
+
"<|as|>",
|
| 102 |
+
"<|tt|>",
|
| 103 |
+
"<|haw|>",
|
| 104 |
+
"<|ln|>",
|
| 105 |
+
"<|ha|>",
|
| 106 |
+
"<|ba|>",
|
| 107 |
+
"<|jw|>",
|
| 108 |
+
"<|su|>",
|
| 109 |
+
"<|yue|>",
|
| 110 |
+
"<|translate|>",
|
| 111 |
+
"<|transcribe|>",
|
| 112 |
+
"<|startoflm|>",
|
| 113 |
+
"<|startofprev|>",
|
| 114 |
+
"<|nospeech|>",
|
| 115 |
+
"<|notimestamps|>"
|
| 116 |
+
],
|
| 117 |
+
"is_local": false,
|
| 118 |
+
"language": null,
|
| 119 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 120 |
+
"pad_token": "<|endoftext|>",
|
| 121 |
+
"predict_timestamps": false,
|
| 122 |
+
"processor_class": "WhisperProcessor",
|
| 123 |
+
"task": null,
|
| 124 |
+
"tokenizer_class": "WhisperTokenizer",
|
| 125 |
+
"unk_token": "<|endoftext|>"
|
| 126 |
+
}
|