mirror of
https://github.com/nad4-su/meeting-minutes.git
synced 2026-10-11 15:50:13 +09:00
feat: dual-stream 캡처로 원격 회의 화자 분리
내 마이크와 시스템(탭) 오디오를 별도 트랙으로 잡고, 트랙마다 인식기를 붙인다. 트랙이 곧 화자이므로 추론도 모델 다운로드도 없이 화자가 갈린다. Chrome이 SpeechRecognition.start(audioTrack)를 지원하는 덕분에 가능하다. 트랙을 명시적으로 넘기므로 기본 마이크를 두고 경합할 일도 없다. - speakers.ts 신설 — 화자 식별자와 표시 이름 해석 - TranscriptChunk.speaker 추가, formatTranscriptChunks가 "화자: 발화"로 출력 - mergeSpeakerChunks — 트랙별 청크를 하나의 시간축으로 병합. 여러 인식기가 같은 timeOrigin을 공유해야 순서가 맞는다 - mergeAdjacentChunks가 화자 경계를 넘어 합치지 않도록 수정 - useSpeechRecognition: audioTrack / speaker / timeOrigin / lang 옵션. 인자 없이 호출하면 기존 동작 그대로다 - useDualStreamCapture 신설 — getUserMedia + getDisplayMedia. video는 즉시 끊고 오디오만 쓴다. 시스템 오디오 확보에 실패하면 마이크 단독으로 폴백하고 화자 분리가 꺼졌음을 안내한다 - LiveRecorder: 화자 분리 토글, 화자 이름 입력, 전사 목록 화자 색상 구분 원격 다자 회의와 대면 회의는 원격 트랙에 diarization이 필요하며 후속 작업이다. 테스트 162 → 176. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
9af00c3016
commit
d2cb5b0009
8 files changed
+546
-19
No files matched your search
@@ -1,8 +1,12 @@
|
||||
import { resolveSpeakerName, type SpeakerNames } from './speakers'
|
||||
|
||||
export interface TranscriptChunk {
|
||||
text: string
|
||||
startTime: number
|
||||
endTime: number
|
||||
isFinal: boolean
|
||||
/** 화자 식별자. 단일 트랙 녹음에서는 없다. */
|
||||
speaker?: string
|
||||
}
|
||||
|
||||
function formatTimestamp(seconds: number): string {
|
||||
@@ -16,14 +20,42 @@ function formatTimestamp(seconds: number): string {
|
||||
return `[${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
|
||||
}
|
||||
|
||||
export function formatTranscriptChunks(chunks: readonly TranscriptChunk[]): string {
|
||||
export function formatTranscriptChunks(
|
||||
chunks: readonly TranscriptChunk[],
|
||||
names?: SpeakerNames,
|
||||
): string {
|
||||
if (chunks.length === 0) return ''
|
||||
|
||||
return chunks
|
||||
.map((chunk) => `${formatTimestamp(chunk.startTime)} ${chunk.text}`)
|
||||
.map((chunk) => {
|
||||
const stamp = formatTimestamp(chunk.startTime)
|
||||
if (!chunk.speaker) return `${stamp} ${chunk.text}`
|
||||
return `${stamp} ${resolveSpeakerName(chunk.speaker, names)}: ${chunk.text}`
|
||||
})
|
||||
.join('\n')
|
||||
}
|
||||
|
||||
/**
|
||||
* 여러 트랙에서 나온 청크를 하나의 시간축으로 합친다.
|
||||
*
|
||||
* 각 인식기가 독립적으로 동작하므로 도착 순서가 발화 순서와 다를 수 있다.
|
||||
* 모든 트랙이 같은 timeOrigin을 공유한다는 전제 아래 startTime으로 정렬한다.
|
||||
*/
|
||||
export function mergeSpeakerChunks(
|
||||
...streams: readonly (readonly TranscriptChunk[])[]
|
||||
): TranscriptChunk[] {
|
||||
const merged = streams.flat()
|
||||
|
||||
return merged
|
||||
.map((chunk, index) => ({ chunk, index }))
|
||||
.sort((a, b) => {
|
||||
const diff = a.chunk.startTime - b.chunk.startTime
|
||||
// 같은 시각이면 원래 순서를 유지해 결과가 흔들리지 않게 한다.
|
||||
return diff !== 0 ? diff : a.index - b.index
|
||||
})
|
||||
.map(({ chunk }) => chunk)
|
||||
}
|
||||
|
||||
export function mergeAdjacentChunks(
|
||||
chunks: readonly TranscriptChunk[],
|
||||
gapThreshold: number,
|
||||
@@ -36,12 +68,19 @@ export function mergeAdjacentChunks(
|
||||
const current = chunks[i]
|
||||
const last = result[result.length - 1]
|
||||
|
||||
// 화자가 다르면 붙이지 않는다 — 서로 다른 사람의 발화가 한 덩어리가 되면 안 된다.
|
||||
if (current.speaker !== last.speaker) {
|
||||
result.push({ ...current })
|
||||
continue
|
||||
}
|
||||
|
||||
if (current.startTime - last.endTime <= gapThreshold) {
|
||||
result[result.length - 1] = {
|
||||
text: `${last.text} ${current.text}`,
|
||||
startTime: last.startTime,
|
||||
endTime: current.endTime,
|
||||
isFinal: current.isFinal,
|
||||
...(last.speaker ? { speaker: last.speaker } : {}),
|
||||
}
|
||||
} else {
|
||||
result.push({ ...current })
|
||||
|
||||
Reference in new issue
Block a user