From d2cb5b0009902e772351f771697319dc875f8c53 Mon Sep 17 00:00:00 2001 From: csbae Date: Mon, 7 Sep 2026 18:27:00 +0900 Subject: [PATCH] =?UTF-8?q?feat:=20dual-stream=20=EC=BA=A1=EC=B2=98?= =?UTF-8?q?=EB=A1=9C=20=EC=9B=90=EA=B2=A9=20=ED=9A=8C=EC=9D=98=20=ED=99=94?= =?UTF-8?q?=EC=9E=90=20=EB=B6=84=EB=A6=AC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 내 마이크와 시스템(탭) 오디오를 별도 트랙으로 잡고, 트랙마다 인식기를 붙인다. 트랙이 곧 화자이므로 추론도 모델 다운로드도 없이 화자가 갈린다. Chrome이 SpeechRecognition.start(audioTrack)를 지원하는 덕분에 가능하다. 트랙을 명시적으로 넘기므로 기본 마이크를 두고 경합할 일도 없다. - speakers.ts 신설 — 화자 식별자와 표시 이름 해석 - TranscriptChunk.speaker 추가, formatTranscriptChunks가 "화자: 발화"로 출력 - mergeSpeakerChunks — 트랙별 청크를 하나의 시간축으로 병합. 여러 인식기가 같은 timeOrigin을 공유해야 순서가 맞는다 - mergeAdjacentChunks가 화자 경계를 넘어 합치지 않도록 수정 - useSpeechRecognition: audioTrack / speaker / timeOrigin / lang 옵션. 인자 없이 호출하면 기존 동작 그대로다 - useDualStreamCapture 신설 — getUserMedia + getDisplayMedia. video는 즉시 끊고 오디오만 쓴다. 시스템 오디오 확보에 실패하면 마이크 단독으로 폴백하고 화자 분리가 꺼졌음을 안내한다 - LiveRecorder: 화자 분리 토글, 화자 이름 입력, 전사 목록 화자 색상 구분 원격 다자 회의와 대면 회의는 원격 트랙에 diarization이 필요하며 후속 작업이다. 테스트 162 → 176. Co-Authored-By: Claude Opus 5 --- src/__tests__/speakers.test.ts | 45 ++++++ src/__tests__/transcript-formatter.test.ts | 81 ++++++++++ src/components/recorder/LiveRecorder.tsx | 178 +++++++++++++++++++-- src/hooks/useDualStreamCapture.ts | 121 ++++++++++++++ src/hooks/useSpeechRecognition.ts | 45 +++++- src/lib/speakers.ts | 46 ++++++ src/lib/transcript-formatter.ts | 43 ++++- src/types/speech-recognition.d.ts | 6 +- 8 files changed, 546 insertions(+), 19 deletions(-) create mode 100644 src/__tests__/speakers.test.ts create mode 100644 src/hooks/useDualStreamCapture.ts create mode 100644 src/lib/speakers.ts diff --git a/src/__tests__/speakers.test.ts b/src/__tests__/speakers.test.ts new file mode 100644 index 0000000..cfc132d --- /dev/null +++ b/src/__tests__/speakers.test.ts @@ -0,0 +1,45 @@ +import { describe, it, expect } from 'vitest' +import { + LOCAL_SPEAKER, + REMOTE_SPEAKER, + listAttendees, + resolveSpeakerName, +} from '@/lib/speakers' + +describe('resolveSpeakerName', () => { + it('기본 화자에 한국어 표기를 붙인다', () => { + expect(resolveSpeakerName(LOCAL_SPEAKER)).toBe('나') + expect(resolveSpeakerName(REMOTE_SPEAKER)).toBe('상대') + }) + + it('사용자가 지정한 이름이 우선한다', () => { + const names = { [REMOTE_SPEAKER]: '김지훈' } + expect(resolveSpeakerName(REMOTE_SPEAKER, names)).toBe('김지훈') + expect(resolveSpeakerName(LOCAL_SPEAKER, names)).toBe('나') + }) + + it('공백뿐인 이름은 무시한다', () => { + expect(resolveSpeakerName(LOCAL_SPEAKER, { [LOCAL_SPEAKER]: ' ' })).toBe('나') + }) + + it('모르는 식별자는 그대로 돌려준다', () => { + expect(resolveSpeakerName('spk_3')).toBe('spk_3') + }) +}) + +describe('listAttendees', () => { + it('중복을 제거하고 순서를 유지한다', () => { + expect( + listAttendees([REMOTE_SPEAKER, LOCAL_SPEAKER, REMOTE_SPEAKER]), + ).toEqual(['상대', '나']) + }) + + it('지정된 이름을 반영한다', () => { + expect( + listAttendees([LOCAL_SPEAKER, REMOTE_SPEAKER], { + [LOCAL_SPEAKER]: '배철승', + [REMOTE_SPEAKER]: '김지훈', + }), + ).toEqual(['배철승', '김지훈']) + }) +}) diff --git a/src/__tests__/transcript-formatter.test.ts b/src/__tests__/transcript-formatter.test.ts index be1739b..681fd4a 100644 --- a/src/__tests__/transcript-formatter.test.ts +++ b/src/__tests__/transcript-formatter.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest' import { formatTranscriptChunks, mergeAdjacentChunks, + mergeSpeakerChunks, type TranscriptChunk, } from '@/lib/transcript-formatter' @@ -55,3 +56,83 @@ describe('mergeAdjacentChunks', () => { expect(result).toEqual([]) }) }) + +describe('화자 표기', () => { + it('speaker가 있으면 이름을 붙여 출력한다', () => { + const out = formatTranscriptChunks([ + { text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'local' }, + { text: '네 반갑습니다', startTime: 3, endTime: 5, isFinal: true, speaker: 'remote' }, + ]) + expect(out).toBe('[00:00] 나: 안녕하세요\n[00:03] 상대: 네 반갑습니다') + }) + + it('지정된 화자 이름을 사용한다', () => { + const out = formatTranscriptChunks( + [{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'remote' }], + { remote: '김지훈' }, + ) + expect(out).toBe('[00:00] 김지훈: 안녕하세요') + }) + + it('speaker가 없으면 기존 형식을 유지한다', () => { + const out = formatTranscriptChunks([ + { text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true }, + ]) + expect(out).toBe('[00:00] 안녕하세요') + }) +}) + +describe('mergeSpeakerChunks', () => { + const local = [ + { text: '먼저 말함', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' }, + { text: '나중에 말함', startTime: 9, endTime: 10, isFinal: true, speaker: 'local' }, + ] + const remote = [ + { text: '중간에 답함', startTime: 4, endTime: 6, isFinal: true, speaker: 'remote' }, + ] + + it('여러 트랙을 시간순으로 합친다', () => { + expect(mergeSpeakerChunks(local, remote).map((c) => c.text)).toEqual([ + '먼저 말함', + '중간에 답함', + '나중에 말함', + ]) + }) + + it('같은 시각이면 넘긴 순서를 유지한다', () => { + const a = [{ text: 'A', startTime: 5, endTime: 6, isFinal: true, speaker: 'local' }] + const b = [{ text: 'B', startTime: 5, endTime: 6, isFinal: true, speaker: 'remote' }] + expect(mergeSpeakerChunks(a, b).map((c) => c.text)).toEqual(['A', 'B']) + }) + + it('빈 트랙을 섞어도 동작한다', () => { + expect(mergeSpeakerChunks([], remote, [])).toEqual(remote) + expect(mergeSpeakerChunks()).toEqual([]) + }) +}) + +describe('mergeAdjacentChunks — 화자 경계', () => { + it('화자가 다르면 간격이 좁아도 합치지 않는다', () => { + const merged = mergeAdjacentChunks( + [ + { text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' }, + { text: '어렵습니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'remote' }, + ], + 5, + ) + expect(merged).toHaveLength(2) + }) + + it('같은 화자면 기존대로 합친다', () => { + const merged = mergeAdjacentChunks( + [ + { text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' }, + { text: '가능합니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' }, + ], + 5, + ) + expect(merged).toHaveLength(1) + expect(merged[0].text).toBe('그건 가능합니다') + expect(merged[0].speaker).toBe('local') + }) +}) diff --git a/src/components/recorder/LiveRecorder.tsx b/src/components/recorder/LiveRecorder.tsx index 9c368dd..9de8bef 100644 --- a/src/components/recorder/LiveRecorder.tsx +++ b/src/components/recorder/LiveRecorder.tsx @@ -1,9 +1,18 @@ 'use client' -import { useEffect, useRef } from 'react' +import { useEffect, useMemo, useRef, useState } from 'react' import { useSpeechRecognition } from '@/hooks/useSpeechRecognition' +import { useDualStreamCapture } from '@/hooks/useDualStreamCapture' import { useLiveSummary } from '@/hooks/useLiveSummary' -import { formatTranscriptChunks } from '@/lib/transcript-formatter' +import { + formatTranscriptChunks, + mergeSpeakerChunks, +} from '@/lib/transcript-formatter' +import { + LOCAL_SPEAKER, + REMOTE_SPEAKER, + resolveSpeakerName, +} from '@/lib/speakers' import type { SummaryDepth, TemplateId } from '@/lib/templates' interface LiveRecorderProps { @@ -21,16 +30,53 @@ export function LiveRecorder({ depth, customPrompt, }: LiveRecorderProps) { + const [speakerSeparation, setSpeakerSeparation] = useState(false) + const [pendingStart, setPendingStart] = useState(false) + const [speakerNames, setSpeakerNames] = useState>({}) + const timeOriginRef = useRef(0) + + const capture = useDualStreamCapture() + + // 화자 분리를 켜면 트랙마다 인식기를 붙인다. 트랙이 곧 화자가 되므로 + // 추론 없이 화자가 갈린다. + const local = useSpeechRecognition({ + audioTrack: speakerSeparation ? capture.micTrack : null, + speaker: speakerSeparation ? LOCAL_SPEAKER : undefined, + timeOrigin: timeOriginRef.current || undefined, + }) + + const remote = useSpeechRecognition({ + audioTrack: capture.systemTrack, + speaker: REMOTE_SPEAKER, + timeOrigin: timeOriginRef.current || undefined, + }) + const { isListening, isSupported, - chunks, interimText, error: recognitionError, startListening, - stopListening, - resetChunks, - } = useSpeechRecognition() + } = local + + const chunks = useMemo( + () => + speakerSeparation + ? mergeSpeakerChunks(local.chunks, remote.chunks) + : local.chunks, + [speakerSeparation, local.chunks, remote.chunks], + ) + + // 캡처가 끝나 트랙이 준비되면 그때 인식기를 시작한다. + useEffect(() => { + if (!pendingStart) return + if (capture.mode === 'idle') return + + setPendingStart(false) + local.startListening() + if (capture.systemTrack) remote.startListening() + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [pendingStart, capture.mode, capture.systemTrack]) const { summary, @@ -78,14 +124,34 @@ export function LiveRecorder({ ) } + async function handleStart() { + timeOriginRef.current = Date.now() + + if (!speakerSeparation) { + startListening() + return + } + + setPendingStart(true) + await capture.start({ withSystemAudio: true }) + } + function handleStop() { - stopListening() - const transcript = formatTranscriptChunks(chunks) + local.stopListening() + remote.stopListening() + capture.stop() + + const transcript = formatTranscriptChunks(chunks, speakerNames) if (transcript.length > 0) { onTranscriptReady(transcript) } } + function handleReset() { + local.resetChunks() + remote.resetChunks() + } + const lastUpdatedLabel = lastUpdatedAt ? new Date(lastUpdatedAt).toLocaleTimeString('ko-KR', { hour: '2-digit', @@ -136,7 +202,7 @@ export function LiveRecorder({ ) : (