diff --git a/src/__tests__/speakers.test.ts b/src/__tests__/speakers.test.ts new file mode 100644 index 0000000..cfc132d --- /dev/null +++ b/src/__tests__/speakers.test.ts @@ -0,0 +1,45 @@ +import { describe, it, expect } from 'vitest' +import { + LOCAL_SPEAKER, + REMOTE_SPEAKER, + listAttendees, + resolveSpeakerName, +} from '@/lib/speakers' + +describe('resolveSpeakerName', () => { + it('기본 화자에 한국어 표기를 붙인다', () => { + expect(resolveSpeakerName(LOCAL_SPEAKER)).toBe('나') + expect(resolveSpeakerName(REMOTE_SPEAKER)).toBe('상대') + }) + + it('사용자가 지정한 이름이 우선한다', () => { + const names = { [REMOTE_SPEAKER]: '김지훈' } + expect(resolveSpeakerName(REMOTE_SPEAKER, names)).toBe('김지훈') + expect(resolveSpeakerName(LOCAL_SPEAKER, names)).toBe('나') + }) + + it('공백뿐인 이름은 무시한다', () => { + expect(resolveSpeakerName(LOCAL_SPEAKER, { [LOCAL_SPEAKER]: ' ' })).toBe('나') + }) + + it('모르는 식별자는 그대로 돌려준다', () => { + expect(resolveSpeakerName('spk_3')).toBe('spk_3') + }) +}) + +describe('listAttendees', () => { + it('중복을 제거하고 순서를 유지한다', () => { + expect( + listAttendees([REMOTE_SPEAKER, LOCAL_SPEAKER, REMOTE_SPEAKER]), + ).toEqual(['상대', '나']) + }) + + it('지정된 이름을 반영한다', () => { + expect( + listAttendees([LOCAL_SPEAKER, REMOTE_SPEAKER], { + [LOCAL_SPEAKER]: '배철승', + [REMOTE_SPEAKER]: '김지훈', + }), + ).toEqual(['배철승', '김지훈']) + }) +}) diff --git a/src/__tests__/transcript-formatter.test.ts b/src/__tests__/transcript-formatter.test.ts index be1739b..681fd4a 100644 --- a/src/__tests__/transcript-formatter.test.ts +++ b/src/__tests__/transcript-formatter.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest' import { formatTranscriptChunks, mergeAdjacentChunks, + mergeSpeakerChunks, type TranscriptChunk, } from '@/lib/transcript-formatter' @@ -55,3 +56,83 @@ describe('mergeAdjacentChunks', () => { expect(result).toEqual([]) }) }) + +describe('화자 표기', () => { + it('speaker가 있으면 이름을 붙여 출력한다', () => { + const out = formatTranscriptChunks([ + { text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'local' }, + { text: '네 반갑습니다', startTime: 3, endTime: 5, isFinal: true, speaker: 'remote' }, + ]) + expect(out).toBe('[00:00] 나: 안녕하세요\n[00:03] 상대: 네 반갑습니다') + }) + + it('지정된 화자 이름을 사용한다', () => { + const out = formatTranscriptChunks( + [{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'remote' }], + { remote: '김지훈' }, + ) + expect(out).toBe('[00:00] 김지훈: 안녕하세요') + }) + + it('speaker가 없으면 기존 형식을 유지한다', () => { + const out = formatTranscriptChunks([ + { text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true }, + ]) + expect(out).toBe('[00:00] 안녕하세요') + }) +}) + +describe('mergeSpeakerChunks', () => { + const local = [ + { text: '먼저 말함', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' }, + { text: '나중에 말함', startTime: 9, endTime: 10, isFinal: true, speaker: 'local' }, + ] + const remote = [ + { text: '중간에 답함', startTime: 4, endTime: 6, isFinal: true, speaker: 'remote' }, + ] + + it('여러 트랙을 시간순으로 합친다', () => { + expect(mergeSpeakerChunks(local, remote).map((c) => c.text)).toEqual([ + '먼저 말함', + '중간에 답함', + '나중에 말함', + ]) + }) + + it('같은 시각이면 넘긴 순서를 유지한다', () => { + const a = [{ text: 'A', startTime: 5, endTime: 6, isFinal: true, speaker: 'local' }] + const b = [{ text: 'B', startTime: 5, endTime: 6, isFinal: true, speaker: 'remote' }] + expect(mergeSpeakerChunks(a, b).map((c) => c.text)).toEqual(['A', 'B']) + }) + + it('빈 트랙을 섞어도 동작한다', () => { + expect(mergeSpeakerChunks([], remote, [])).toEqual(remote) + expect(mergeSpeakerChunks()).toEqual([]) + }) +}) + +describe('mergeAdjacentChunks — 화자 경계', () => { + it('화자가 다르면 간격이 좁아도 합치지 않는다', () => { + const merged = mergeAdjacentChunks( + [ + { text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' }, + { text: '어렵습니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'remote' }, + ], + 5, + ) + expect(merged).toHaveLength(2) + }) + + it('같은 화자면 기존대로 합친다', () => { + const merged = mergeAdjacentChunks( + [ + { text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' }, + { text: '가능합니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' }, + ], + 5, + ) + expect(merged).toHaveLength(1) + expect(merged[0].text).toBe('그건 가능합니다') + expect(merged[0].speaker).toBe('local') + }) +}) diff --git a/src/components/recorder/LiveRecorder.tsx b/src/components/recorder/LiveRecorder.tsx index 9c368dd..9de8bef 100644 --- a/src/components/recorder/LiveRecorder.tsx +++ b/src/components/recorder/LiveRecorder.tsx @@ -1,9 +1,18 @@ 'use client' -import { useEffect, useRef } from 'react' +import { useEffect, useMemo, useRef, useState } from 'react' import { useSpeechRecognition } from '@/hooks/useSpeechRecognition' +import { useDualStreamCapture } from '@/hooks/useDualStreamCapture' import { useLiveSummary } from '@/hooks/useLiveSummary' -import { formatTranscriptChunks } from '@/lib/transcript-formatter' +import { + formatTranscriptChunks, + mergeSpeakerChunks, +} from '@/lib/transcript-formatter' +import { + LOCAL_SPEAKER, + REMOTE_SPEAKER, + resolveSpeakerName, +} from '@/lib/speakers' import type { SummaryDepth, TemplateId } from '@/lib/templates' interface LiveRecorderProps { @@ -21,16 +30,53 @@ export function LiveRecorder({ depth, customPrompt, }: LiveRecorderProps) { + const [speakerSeparation, setSpeakerSeparation] = useState(false) + const [pendingStart, setPendingStart] = useState(false) + const [speakerNames, setSpeakerNames] = useState>({}) + const timeOriginRef = useRef(0) + + const capture = useDualStreamCapture() + + // 화자 분리를 켜면 트랙마다 인식기를 붙인다. 트랙이 곧 화자가 되므로 + // 추론 없이 화자가 갈린다. + const local = useSpeechRecognition({ + audioTrack: speakerSeparation ? capture.micTrack : null, + speaker: speakerSeparation ? LOCAL_SPEAKER : undefined, + timeOrigin: timeOriginRef.current || undefined, + }) + + const remote = useSpeechRecognition({ + audioTrack: capture.systemTrack, + speaker: REMOTE_SPEAKER, + timeOrigin: timeOriginRef.current || undefined, + }) + const { isListening, isSupported, - chunks, interimText, error: recognitionError, startListening, - stopListening, - resetChunks, - } = useSpeechRecognition() + } = local + + const chunks = useMemo( + () => + speakerSeparation + ? mergeSpeakerChunks(local.chunks, remote.chunks) + : local.chunks, + [speakerSeparation, local.chunks, remote.chunks], + ) + + // 캡처가 끝나 트랙이 준비되면 그때 인식기를 시작한다. + useEffect(() => { + if (!pendingStart) return + if (capture.mode === 'idle') return + + setPendingStart(false) + local.startListening() + if (capture.systemTrack) remote.startListening() + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [pendingStart, capture.mode, capture.systemTrack]) const { summary, @@ -78,14 +124,34 @@ export function LiveRecorder({ ) } + async function handleStart() { + timeOriginRef.current = Date.now() + + if (!speakerSeparation) { + startListening() + return + } + + setPendingStart(true) + await capture.start({ withSystemAudio: true }) + } + function handleStop() { - stopListening() - const transcript = formatTranscriptChunks(chunks) + local.stopListening() + remote.stopListening() + capture.stop() + + const transcript = formatTranscriptChunks(chunks, speakerNames) if (transcript.length > 0) { onTranscriptReady(transcript) } } + function handleReset() { + local.resetChunks() + remote.resetChunks() + } + const lastUpdatedLabel = lastUpdatedAt ? new Date(lastUpdatedAt).toLocaleTimeString('ko-KR', { hour: '2-digit', @@ -136,7 +202,7 @@ export function LiveRecorder({ ) : (