VAC activated by default
This commit is contained in:
parent
e2184d5e06
commit
e42523af84
2 changed files with 10 additions and 17 deletions
|
|
@ -78,9 +78,6 @@ brew install ffmpeg
|
||||||
### Optional Dependencies
|
### Optional Dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Voice Activity Controller (prevents hallucinations)
|
|
||||||
pip install torch
|
|
||||||
|
|
||||||
# Sentence-based buffer trimming
|
# Sentence-based buffer trimming
|
||||||
pip install mosestokenizer wtpsplit
|
pip install mosestokenizer wtpsplit
|
||||||
pip install tokenize_uk # If you work with Ukrainian text
|
pip install tokenize_uk # If you work with Ukrainian text
|
||||||
|
|
@ -93,7 +90,6 @@ pip install whisperlivekit[whisper] # Original Whisper
|
||||||
pip install whisperlivekit[whisper-timestamped] # Improved timestamps
|
pip install whisperlivekit[whisper-timestamped] # Improved timestamps
|
||||||
pip install whisperlivekit[mlx-whisper] # Apple Silicon optimization
|
pip install whisperlivekit[mlx-whisper] # Apple Silicon optimization
|
||||||
pip install whisperlivekit[openai] # OpenAI API
|
pip install whisperlivekit[openai] # OpenAI API
|
||||||
pip install whisperlivekit[simulstreaming]
|
|
||||||
```
|
```
|
||||||
|
|
||||||
### 🎹 Pyannote Models Setup
|
### 🎹 Pyannote Models Setup
|
||||||
|
|
@ -195,7 +191,7 @@ WhisperLiveKit offers extensive configuration options:
|
||||||
| `--punctuation-split` | Use punctuation to improve speaker boundaries | `True` |
|
| `--punctuation-split` | Use punctuation to improve speaker boundaries | `True` |
|
||||||
| `--confidence-validation` | Use confidence scores for faster validation | `False` |
|
| `--confidence-validation` | Use confidence scores for faster validation | `False` |
|
||||||
| `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` |
|
| `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` |
|
||||||
| `--vac` | Use Voice Activity Controller | `False` |
|
| `--vac` | Use Voice Activity Controller | `True` |
|
||||||
| `--no-vad` | Disable Voice Activity Detection | `False` |
|
| `--no-vad` | Disable Voice Activity Detection | `False` |
|
||||||
| `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` |
|
| `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` |
|
||||||
| `--warmup-file` | Audio file path for model warmup | `jfk.wav` |
|
| `--warmup-file` | Audio file path for model warmup | `jfk.wav` |
|
||||||
|
|
|
||||||
|
|
@ -27,24 +27,21 @@ dependencies = [
|
||||||
"soundfile",
|
"soundfile",
|
||||||
"faster-whisper",
|
"faster-whisper",
|
||||||
"uvicorn",
|
"uvicorn",
|
||||||
"websockets"
|
"websockets",
|
||||||
]
|
|
||||||
|
|
||||||
[project.optional-dependencies]
|
|
||||||
diarization = ["diart"]
|
|
||||||
vac = ["torch"]
|
|
||||||
sentence = ["mosestokenizer", "wtpsplit"]
|
|
||||||
whisper = ["whisper"]
|
|
||||||
whisper-timestamped = ["whisper-timestamped"]
|
|
||||||
mlx-whisper = ["mlx-whisper"]
|
|
||||||
openai = ["openai"]
|
|
||||||
simulstreaming = [
|
|
||||||
"torch",
|
"torch",
|
||||||
"tqdm",
|
"tqdm",
|
||||||
"tiktoken",
|
"tiktoken",
|
||||||
'triton>=2.0.0,<3; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")'
|
'triton>=2.0.0,<3; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")'
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[project.optional-dependencies]
|
||||||
|
diarization = ["diart"]
|
||||||
|
sentence = ["mosestokenizer", "wtpsplit"]
|
||||||
|
whisper = ["whisper"]
|
||||||
|
whisper-timestamped = ["whisper-timestamped"]
|
||||||
|
mlx-whisper = ["mlx-whisper"]
|
||||||
|
openai = ["openai"]
|
||||||
|
|
||||||
[project.urls]
|
[project.urls]
|
||||||
Homepage = "https://github.com/QuentinFuxa/WhisperLiveKit"
|
Homepage = "https://github.com/QuentinFuxa/WhisperLiveKit"
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue