Improve speaker handling; update sleep duration and manage speaker transitions more effectively
This commit is contained in:
parent
58eba2a1f6
commit
2608abf0f3
1 changed files with 16 additions and 15 deletions
|
|
@ -214,10 +214,10 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||||
else:
|
else:
|
||||||
chunk_history.append({
|
chunk_history.append({
|
||||||
"beg": time() - beg_loop,
|
"beg": time() - beg_loop,
|
||||||
"end": time() - beg_loop + 0.1,
|
"end": time() - beg_loop + 1,
|
||||||
"text": '',
|
"text": '',
|
||||||
})
|
})
|
||||||
sleep(0.1)
|
sleep(1)
|
||||||
buffer = ''
|
buffer = ''
|
||||||
|
|
||||||
if args.diarization:
|
if args.diarization:
|
||||||
|
|
@ -225,28 +225,29 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||||
diarization.assign_speakers_to_chunks(chunk_history)
|
diarization.assign_speakers_to_chunks(chunk_history)
|
||||||
|
|
||||||
|
|
||||||
current_speaker = -1
|
current_speaker = 0
|
||||||
lines = [{
|
lines = []
|
||||||
"beg": 0,
|
last_end_diarized = 0
|
||||||
"end": 0,
|
for ind, ch in enumerate(chunk_history):
|
||||||
"speaker": current_speaker,
|
speaker = ch.get("speaker", -3)
|
||||||
"text": ""
|
if speaker == -1 and ind < len(chunk_history) - 1:
|
||||||
}]
|
continue
|
||||||
for ch in chunk_history:
|
elif speaker != current_speaker:
|
||||||
if args.diarization and ch["speaker"] and ch["speaker"] != current_speaker:
|
|
||||||
new_speaker = ch["speaker"]
|
|
||||||
lines.append(
|
lines.append(
|
||||||
{
|
{
|
||||||
"speaker": new_speaker,
|
"speaker": speaker,
|
||||||
"text": ch['text'],
|
"text": ch['text'],
|
||||||
"beg": format_time(ch['beg']),
|
"beg": format_time(ch['beg']),
|
||||||
"end": format_time(ch['end']),
|
"end": format_time(ch['end']),
|
||||||
|
"diff": round(ch['end'] - last_end_diarized, 2)
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
current_speaker = new_speaker
|
current_speaker = speaker
|
||||||
else:
|
elif speaker != -1:
|
||||||
lines[-1]["text"] += ch['text']
|
lines[-1]["text"] += ch['text']
|
||||||
lines[-1]["end"] = format_time(ch['end'])
|
lines[-1]["end"] = format_time(ch['end'])
|
||||||
|
if speaker != -1:
|
||||||
|
last_end_diarized = max(ch['end'], last_end_diarized)
|
||||||
|
|
||||||
response = {"lines": lines, "buffer": buffer}
|
response = {"lines": lines, "buffer": buffer}
|
||||||
await websocket.send_json(response)
|
await websocket.send_json(response)
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue