Enhance diarization logic to improve speaker attribution : corrects several bugs

This commit is contained in:
Quentin Fuxa 2025-02-23 23:14:10 +01:00
parent 34b707d84e
commit 296327071d

View file

@ -208,6 +208,7 @@ async def websocket_endpoint(websocket: WebSocket):
"beg": transcription.start, "beg": transcription.start,
"end": transcription.end, "end": transcription.end,
"text": transcription.text, "text": transcription.text,
"speaker": -1
}) })
full_transcription += transcription.text if transcription else "" full_transcription += transcription.text if transcription else ""
buffer = online.get_buffer() buffer = online.get_buffer()
@ -218,23 +219,32 @@ async def websocket_endpoint(websocket: WebSocket):
"beg": time() - beg_loop, "beg": time() - beg_loop,
"end": time() - beg_loop + 1, "end": time() - beg_loop + 1,
"text": '', "text": '',
"speaker": -1
}) })
sleep(1) sleep(1)
buffer = '' buffer = ''
if args.diarization: if args.diarization:
await diarization.diarize(pcm_array) await diarization.diarize(pcm_array)
diarization.assign_speakers_to_chunks(chunk_history) end_attributed_speaker = diarization.assign_speakers_to_chunks(chunk_history)
current_speaker = 0 current_speaker = -10
lines = [] lines = []
last_end_diarized = 0 last_end_diarized = 0
previous_speaker = -1
for ind, ch in enumerate(chunk_history): for ind, ch in enumerate(chunk_history):
speaker = ch.get("speaker", -3) speaker = ch.get("speaker")
if speaker == -1 and ind < len(chunk_history) - 1: if args.diarization:
continue if speaker == -1 or speaker == 0:
elif speaker != current_speaker: if ch['end'] < end_attributed_speaker:
speaker = previous_speaker
else:
speaker = 0
else:
last_end_diarized = max(ch['end'], last_end_diarized)
if speaker != current_speaker:
lines.append( lines.append(
{ {
"speaker": speaker, "speaker": speaker,
@ -245,11 +255,10 @@ async def websocket_endpoint(websocket: WebSocket):
} }
) )
current_speaker = speaker current_speaker = speaker
elif speaker != -1: else:
lines[-1]["text"] += ch['text'] lines[-1]["text"] += ch['text']
lines[-1]["end"] = format_time(ch['end']) lines[-1]["end"] = format_time(ch['end'])
if speaker != -1: lines[-1]["diff"] = round(ch['end'] - last_end_diarized, 2)
last_end_diarized = max(ch['end'], last_end_diarized)
response = {"lines": lines, "buffer": buffer} response = {"lines": lines, "buffer": buffer}
await websocket.send_json(response) await websocket.send_json(response)