VAC activated by default

This commit is contained in:
Quentin Fuxa 2025-08-17 01:29:34 +02:00
parent e2184d5e06
commit e42523af84
2 changed files with 10 additions and 17 deletions

View file

@ -78,9 +78,6 @@ brew install ffmpeg
### Optional Dependencies ### Optional Dependencies
```bash ```bash
# Voice Activity Controller (prevents hallucinations)
pip install torch
# Sentence-based buffer trimming # Sentence-based buffer trimming
pip install mosestokenizer wtpsplit pip install mosestokenizer wtpsplit
pip install tokenize_uk # If you work with Ukrainian text pip install tokenize_uk # If you work with Ukrainian text
@ -93,7 +90,6 @@ pip install whisperlivekit[whisper] # Original Whisper
pip install whisperlivekit[whisper-timestamped] # Improved timestamps pip install whisperlivekit[whisper-timestamped] # Improved timestamps
pip install whisperlivekit[mlx-whisper] # Apple Silicon optimization pip install whisperlivekit[mlx-whisper] # Apple Silicon optimization
pip install whisperlivekit[openai] # OpenAI API pip install whisperlivekit[openai] # OpenAI API
pip install whisperlivekit[simulstreaming]
``` ```
### 🎹 Pyannote Models Setup ### 🎹 Pyannote Models Setup
@ -195,7 +191,7 @@ WhisperLiveKit offers extensive configuration options:
| `--punctuation-split` | Use punctuation to improve speaker boundaries | `True` | | `--punctuation-split` | Use punctuation to improve speaker boundaries | `True` |
| `--confidence-validation` | Use confidence scores for faster validation | `False` | | `--confidence-validation` | Use confidence scores for faster validation | `False` |
| `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` | | `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` |
| `--vac` | Use Voice Activity Controller | `False` | | `--vac` | Use Voice Activity Controller | `True` |
| `--no-vad` | Disable Voice Activity Detection | `False` | | `--no-vad` | Disable Voice Activity Detection | `False` |
| `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` | | `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` |
| `--warmup-file` | Audio file path for model warmup | `jfk.wav` | | `--warmup-file` | Audio file path for model warmup | `jfk.wav` |

View file

@ -27,24 +27,21 @@ dependencies = [
"soundfile", "soundfile",
"faster-whisper", "faster-whisper",
"uvicorn", "uvicorn",
"websockets" "websockets",
]
[project.optional-dependencies]
diarization = ["diart"]
vac = ["torch"]
sentence = ["mosestokenizer", "wtpsplit"]
whisper = ["whisper"]
whisper-timestamped = ["whisper-timestamped"]
mlx-whisper = ["mlx-whisper"]
openai = ["openai"]
simulstreaming = [
"torch", "torch",
"tqdm", "tqdm",
"tiktoken", "tiktoken",
'triton>=2.0.0,<3; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")' 'triton>=2.0.0,<3; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")'
] ]
[project.optional-dependencies]
diarization = ["diart"]
sentence = ["mosestokenizer", "wtpsplit"]
whisper = ["whisper"]
whisper-timestamped = ["whisper-timestamped"]
mlx-whisper = ["mlx-whisper"]
openai = ["openai"]
[project.urls] [project.urls]
Homepage = "https://github.com/QuentinFuxa/WhisperLiveKit" Homepage = "https://github.com/QuentinFuxa/WhisperLiveKit"