mirror of
https://github.com/nad4-su/meeting-minutes.git
synced 2026-10-11 15:50:13 +09:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a8cfab8a4a | ||
|
|
d2cb5b0009 |
No files matched your search
@@ -0,0 +1,45 @@
|
|||||||
|
import { describe, it, expect } from 'vitest'
|
||||||
|
import {
|
||||||
|
LOCAL_SPEAKER,
|
||||||
|
REMOTE_SPEAKER,
|
||||||
|
listAttendees,
|
||||||
|
resolveSpeakerName,
|
||||||
|
} from '@/lib/speakers'
|
||||||
|
|
||||||
|
describe('resolveSpeakerName', () => {
|
||||||
|
it('기본 화자에 한국어 표기를 붙인다', () => {
|
||||||
|
expect(resolveSpeakerName(LOCAL_SPEAKER)).toBe('나')
|
||||||
|
expect(resolveSpeakerName(REMOTE_SPEAKER)).toBe('상대')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('사용자가 지정한 이름이 우선한다', () => {
|
||||||
|
const names = { [REMOTE_SPEAKER]: '김지훈' }
|
||||||
|
expect(resolveSpeakerName(REMOTE_SPEAKER, names)).toBe('김지훈')
|
||||||
|
expect(resolveSpeakerName(LOCAL_SPEAKER, names)).toBe('나')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('공백뿐인 이름은 무시한다', () => {
|
||||||
|
expect(resolveSpeakerName(LOCAL_SPEAKER, { [LOCAL_SPEAKER]: ' ' })).toBe('나')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('모르는 식별자는 그대로 돌려준다', () => {
|
||||||
|
expect(resolveSpeakerName('spk_3')).toBe('spk_3')
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('listAttendees', () => {
|
||||||
|
it('중복을 제거하고 순서를 유지한다', () => {
|
||||||
|
expect(
|
||||||
|
listAttendees([REMOTE_SPEAKER, LOCAL_SPEAKER, REMOTE_SPEAKER]),
|
||||||
|
).toEqual(['상대', '나'])
|
||||||
|
})
|
||||||
|
|
||||||
|
it('지정된 이름을 반영한다', () => {
|
||||||
|
expect(
|
||||||
|
listAttendees([LOCAL_SPEAKER, REMOTE_SPEAKER], {
|
||||||
|
[LOCAL_SPEAKER]: '배철승',
|
||||||
|
[REMOTE_SPEAKER]: '김지훈',
|
||||||
|
}),
|
||||||
|
).toEqual(['배철승', '김지훈'])
|
||||||
|
})
|
||||||
|
})
|
||||||
@@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest'
|
|||||||
import {
|
import {
|
||||||
formatTranscriptChunks,
|
formatTranscriptChunks,
|
||||||
mergeAdjacentChunks,
|
mergeAdjacentChunks,
|
||||||
|
mergeSpeakerChunks,
|
||||||
type TranscriptChunk,
|
type TranscriptChunk,
|
||||||
} from '@/lib/transcript-formatter'
|
} from '@/lib/transcript-formatter'
|
||||||
|
|
||||||
@@ -55,3 +56,83 @@ describe('mergeAdjacentChunks', () => {
|
|||||||
expect(result).toEqual([])
|
expect(result).toEqual([])
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
describe('화자 표기', () => {
|
||||||
|
it('speaker가 있으면 이름을 붙여 출력한다', () => {
|
||||||
|
const out = formatTranscriptChunks([
|
||||||
|
{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'local' },
|
||||||
|
{ text: '네 반갑습니다', startTime: 3, endTime: 5, isFinal: true, speaker: 'remote' },
|
||||||
|
])
|
||||||
|
expect(out).toBe('[00:00] 나: 안녕하세요\n[00:03] 상대: 네 반갑습니다')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('지정된 화자 이름을 사용한다', () => {
|
||||||
|
const out = formatTranscriptChunks(
|
||||||
|
[{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'remote' }],
|
||||||
|
{ remote: '김지훈' },
|
||||||
|
)
|
||||||
|
expect(out).toBe('[00:00] 김지훈: 안녕하세요')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('speaker가 없으면 기존 형식을 유지한다', () => {
|
||||||
|
const out = formatTranscriptChunks([
|
||||||
|
{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true },
|
||||||
|
])
|
||||||
|
expect(out).toBe('[00:00] 안녕하세요')
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('mergeSpeakerChunks', () => {
|
||||||
|
const local = [
|
||||||
|
{ text: '먼저 말함', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' },
|
||||||
|
{ text: '나중에 말함', startTime: 9, endTime: 10, isFinal: true, speaker: 'local' },
|
||||||
|
]
|
||||||
|
const remote = [
|
||||||
|
{ text: '중간에 답함', startTime: 4, endTime: 6, isFinal: true, speaker: 'remote' },
|
||||||
|
]
|
||||||
|
|
||||||
|
it('여러 트랙을 시간순으로 합친다', () => {
|
||||||
|
expect(mergeSpeakerChunks(local, remote).map((c) => c.text)).toEqual([
|
||||||
|
'먼저 말함',
|
||||||
|
'중간에 답함',
|
||||||
|
'나중에 말함',
|
||||||
|
])
|
||||||
|
})
|
||||||
|
|
||||||
|
it('같은 시각이면 넘긴 순서를 유지한다', () => {
|
||||||
|
const a = [{ text: 'A', startTime: 5, endTime: 6, isFinal: true, speaker: 'local' }]
|
||||||
|
const b = [{ text: 'B', startTime: 5, endTime: 6, isFinal: true, speaker: 'remote' }]
|
||||||
|
expect(mergeSpeakerChunks(a, b).map((c) => c.text)).toEqual(['A', 'B'])
|
||||||
|
})
|
||||||
|
|
||||||
|
it('빈 트랙을 섞어도 동작한다', () => {
|
||||||
|
expect(mergeSpeakerChunks([], remote, [])).toEqual(remote)
|
||||||
|
expect(mergeSpeakerChunks()).toEqual([])
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('mergeAdjacentChunks — 화자 경계', () => {
|
||||||
|
it('화자가 다르면 간격이 좁아도 합치지 않는다', () => {
|
||||||
|
const merged = mergeAdjacentChunks(
|
||||||
|
[
|
||||||
|
{ text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' },
|
||||||
|
{ text: '어렵습니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'remote' },
|
||||||
|
],
|
||||||
|
5,
|
||||||
|
)
|
||||||
|
expect(merged).toHaveLength(2)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('같은 화자면 기존대로 합친다', () => {
|
||||||
|
const merged = mergeAdjacentChunks(
|
||||||
|
[
|
||||||
|
{ text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' },
|
||||||
|
{ text: '가능합니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' },
|
||||||
|
],
|
||||||
|
5,
|
||||||
|
)
|
||||||
|
expect(merged).toHaveLength(1)
|
||||||
|
expect(merged[0].text).toBe('그건 가능합니다')
|
||||||
|
expect(merged[0].speaker).toBe('local')
|
||||||
|
})
|
||||||
|
})
|
||||||
@@ -1,9 +1,18 @@
|
|||||||
'use client'
|
'use client'
|
||||||
|
|
||||||
import { useEffect, useRef } from 'react'
|
import { useEffect, useMemo, useRef, useState } from 'react'
|
||||||
import { useSpeechRecognition } from '@/hooks/useSpeechRecognition'
|
import { useSpeechRecognition } from '@/hooks/useSpeechRecognition'
|
||||||
|
import { useDualStreamCapture } from '@/hooks/useDualStreamCapture'
|
||||||
import { useLiveSummary } from '@/hooks/useLiveSummary'
|
import { useLiveSummary } from '@/hooks/useLiveSummary'
|
||||||
import { formatTranscriptChunks } from '@/lib/transcript-formatter'
|
import {
|
||||||
|
formatTranscriptChunks,
|
||||||
|
mergeSpeakerChunks,
|
||||||
|
} from '@/lib/transcript-formatter'
|
||||||
|
import {
|
||||||
|
LOCAL_SPEAKER,
|
||||||
|
REMOTE_SPEAKER,
|
||||||
|
resolveSpeakerName,
|
||||||
|
} from '@/lib/speakers'
|
||||||
import type { SummaryDepth, TemplateId } from '@/lib/templates'
|
import type { SummaryDepth, TemplateId } from '@/lib/templates'
|
||||||
|
|
||||||
interface LiveRecorderProps {
|
interface LiveRecorderProps {
|
||||||
@@ -21,16 +30,57 @@ export function LiveRecorder({
|
|||||||
depth,
|
depth,
|
||||||
customPrompt,
|
customPrompt,
|
||||||
}: LiveRecorderProps) {
|
}: LiveRecorderProps) {
|
||||||
|
const [speakerSeparation, setSpeakerSeparation] = useState(false)
|
||||||
|
const [pendingStart, setPendingStart] = useState(false)
|
||||||
|
const [speakerNames, setSpeakerNames] = useState<Record<string, string>>({})
|
||||||
|
const timeOriginRef = useRef(0)
|
||||||
|
|
||||||
|
const capture = useDualStreamCapture()
|
||||||
|
|
||||||
|
// 화자 분리를 켜면 트랙마다 인식기를 붙인다. 트랙이 곧 화자가 되므로
|
||||||
|
// 추론 없이 화자가 갈린다.
|
||||||
|
// 두 트랙을 실제로 확보했을 때만 화자를 라벨링한다.
|
||||||
|
// 마이크만 잡힌 상태에서 라벨을 붙이면 상대 발언이 내 것으로 기록된다.
|
||||||
|
const isSeparating = capture.mode === 'dual-stream'
|
||||||
|
|
||||||
|
const local = useSpeechRecognition({
|
||||||
|
audioTrack: speakerSeparation ? capture.micTrack : null,
|
||||||
|
speaker: isSeparating ? LOCAL_SPEAKER : undefined,
|
||||||
|
timeOrigin: timeOriginRef.current || undefined,
|
||||||
|
})
|
||||||
|
|
||||||
|
const remote = useSpeechRecognition({
|
||||||
|
audioTrack: capture.systemTrack,
|
||||||
|
speaker: REMOTE_SPEAKER,
|
||||||
|
timeOrigin: timeOriginRef.current || undefined,
|
||||||
|
})
|
||||||
|
|
||||||
const {
|
const {
|
||||||
isListening,
|
isListening,
|
||||||
isSupported,
|
isSupported,
|
||||||
chunks,
|
|
||||||
interimText,
|
interimText,
|
||||||
error: recognitionError,
|
error: recognitionError,
|
||||||
startListening,
|
startListening,
|
||||||
stopListening,
|
} = local
|
||||||
resetChunks,
|
|
||||||
} = useSpeechRecognition()
|
const chunks = useMemo(
|
||||||
|
() =>
|
||||||
|
isSeparating
|
||||||
|
? mergeSpeakerChunks(local.chunks, remote.chunks)
|
||||||
|
: local.chunks,
|
||||||
|
[isSeparating, local.chunks, remote.chunks],
|
||||||
|
)
|
||||||
|
|
||||||
|
// 캡처가 끝나 트랙이 준비되면 그때 인식기를 시작한다.
|
||||||
|
useEffect(() => {
|
||||||
|
if (!pendingStart) return
|
||||||
|
if (capture.mode === 'idle') return
|
||||||
|
|
||||||
|
setPendingStart(false)
|
||||||
|
local.startListening()
|
||||||
|
if (capture.systemTrack) remote.startListening()
|
||||||
|
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||||
|
}, [pendingStart, capture.mode, capture.systemTrack])
|
||||||
|
|
||||||
const {
|
const {
|
||||||
summary,
|
summary,
|
||||||
@@ -78,14 +128,34 @@ export function LiveRecorder({
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async function handleStart() {
|
||||||
|
timeOriginRef.current = Date.now()
|
||||||
|
|
||||||
|
if (!speakerSeparation) {
|
||||||
|
startListening()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
setPendingStart(true)
|
||||||
|
await capture.start({ withSystemAudio: true })
|
||||||
|
}
|
||||||
|
|
||||||
function handleStop() {
|
function handleStop() {
|
||||||
stopListening()
|
local.stopListening()
|
||||||
const transcript = formatTranscriptChunks(chunks)
|
remote.stopListening()
|
||||||
|
capture.stop()
|
||||||
|
|
||||||
|
const transcript = formatTranscriptChunks(chunks, speakerNames)
|
||||||
if (transcript.length > 0) {
|
if (transcript.length > 0) {
|
||||||
onTranscriptReady(transcript)
|
onTranscriptReady(transcript)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function handleReset() {
|
||||||
|
local.resetChunks()
|
||||||
|
remote.resetChunks()
|
||||||
|
}
|
||||||
|
|
||||||
const lastUpdatedLabel = lastUpdatedAt
|
const lastUpdatedLabel = lastUpdatedAt
|
||||||
? new Date(lastUpdatedAt).toLocaleTimeString('ko-KR', {
|
? new Date(lastUpdatedAt).toLocaleTimeString('ko-KR', {
|
||||||
hour: '2-digit',
|
hour: '2-digit',
|
||||||
@@ -136,7 +206,7 @@ export function LiveRecorder({
|
|||||||
</button>
|
</button>
|
||||||
) : (
|
) : (
|
||||||
<button
|
<button
|
||||||
onClick={startListening}
|
onClick={handleStart}
|
||||||
className="flex items-center gap-2 rounded-full bg-blue-600 px-6 py-3 text-white font-medium shadow-lg shadow-blue-600/25 hover:bg-blue-700 transition-colors"
|
className="flex items-center gap-2 rounded-full bg-blue-600 px-6 py-3 text-white font-medium shadow-lg shadow-blue-600/25 hover:bg-blue-700 transition-colors"
|
||||||
>
|
>
|
||||||
<span className="text-lg">🎤</span>
|
<span className="text-lg">🎤</span>
|
||||||
@@ -146,7 +216,7 @@ export function LiveRecorder({
|
|||||||
|
|
||||||
{chunks.length > 0 && !isListening && (
|
{chunks.length > 0 && !isListening && (
|
||||||
<button
|
<button
|
||||||
onClick={resetChunks}
|
onClick={handleReset}
|
||||||
className="rounded-full border border-neutral-300 px-4 py-2 text-sm text-neutral-600 hover:bg-neutral-50 transition-colors"
|
className="rounded-full border border-neutral-300 px-4 py-2 text-sm text-neutral-600 hover:bg-neutral-50 transition-colors"
|
||||||
>
|
>
|
||||||
초기화
|
초기화
|
||||||
@@ -159,7 +229,89 @@ export function LiveRecorder({
|
|||||||
녹음 중 · {wordCount}단어 · {chunks.length}개 구간
|
녹음 중 · {wordCount}단어 · {chunks.length}개 구간
|
||||||
</div>
|
</div>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
{isListening && capture.mode === 'dual-stream' && (
|
||||||
|
<div className="flex items-center gap-1.5 rounded-full bg-purple-50 px-4 py-1.5 text-xs text-purple-700">
|
||||||
|
<span className="h-2 w-2 rounded-full bg-purple-500" />
|
||||||
|
화자 분리 중
|
||||||
</div>
|
</div>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
|
||||||
|
{!isListening && chunks.length === 0 && (
|
||||||
|
<div className="rounded-xl border border-neutral-200 bg-white p-4">
|
||||||
|
<label className="flex items-start gap-3 cursor-pointer">
|
||||||
|
<input
|
||||||
|
type="checkbox"
|
||||||
|
checked={speakerSeparation}
|
||||||
|
onChange={(e) => setSpeakerSeparation(e.target.checked)}
|
||||||
|
className="mt-0.5 h-4 w-4 accent-purple-600"
|
||||||
|
/>
|
||||||
|
<span>
|
||||||
|
<span className="text-sm font-medium text-neutral-800">
|
||||||
|
화자 분리 (원격 회의)
|
||||||
|
</span>
|
||||||
|
<span className="mt-0.5 block text-xs text-neutral-500">
|
||||||
|
내 마이크와 상대 목소리를 별도 트랙으로 받아 구분합니다. 추론이 없어
|
||||||
|
정확합니다. 시작 시 <strong>화면 공유 대화상자에서 “탭 오디오 공유”를
|
||||||
|
반드시 켜주세요.</strong>
|
||||||
|
</span>
|
||||||
|
<span className="mt-1.5 block text-xs text-amber-700">
|
||||||
|
한 대의 노트북을 앞에 두고 마주 앉은 <strong>대면 회의에서는 동작하지
|
||||||
|
않습니다.</strong> 마이크 하나에 두 사람 목소리가 섞여 들어오기 때문입니다.
|
||||||
|
이 경우 화자 라벨 없이 녹음되며, 대면 화자 분리는 준비 중입니다.
|
||||||
|
</span>
|
||||||
|
</span>
|
||||||
|
</label>
|
||||||
|
|
||||||
|
{speakerSeparation && (
|
||||||
|
<div className="mt-4 grid gap-3 sm:grid-cols-2">
|
||||||
|
<label className="block">
|
||||||
|
<span className="mb-1 block text-xs text-neutral-500">내 이름</span>
|
||||||
|
<input
|
||||||
|
type="text"
|
||||||
|
value={speakerNames[LOCAL_SPEAKER] ?? ''}
|
||||||
|
onChange={(e) =>
|
||||||
|
setSpeakerNames((prev) => ({
|
||||||
|
...prev,
|
||||||
|
[LOCAL_SPEAKER]: e.target.value,
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
placeholder="나"
|
||||||
|
className="w-full rounded-lg border border-neutral-300 px-3 py-2 text-sm focus:border-purple-400 focus:outline-none"
|
||||||
|
/>
|
||||||
|
</label>
|
||||||
|
<label className="block">
|
||||||
|
<span className="mb-1 block text-xs text-neutral-500">상대 이름</span>
|
||||||
|
<input
|
||||||
|
type="text"
|
||||||
|
value={speakerNames[REMOTE_SPEAKER] ?? ''}
|
||||||
|
onChange={(e) =>
|
||||||
|
setSpeakerNames((prev) => ({
|
||||||
|
...prev,
|
||||||
|
[REMOTE_SPEAKER]: e.target.value,
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
placeholder="상대"
|
||||||
|
className="w-full rounded-lg border border-neutral-300 px-3 py-2 text-sm focus:border-purple-400 focus:outline-none"
|
||||||
|
/>
|
||||||
|
</label>
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
|
{capture.error && (
|
||||||
|
<div className="rounded-xl border border-red-200 bg-red-50 p-4 text-sm text-red-700">
|
||||||
|
<strong>오디오 캡처 오류:</strong> {capture.error}
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
|
{capture.warning && (
|
||||||
|
<div className="rounded-xl border border-amber-200 bg-amber-50 p-4 text-sm text-amber-800">
|
||||||
|
⚠️ {capture.warning}
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
{recognitionError && (
|
{recognitionError && (
|
||||||
<div className="rounded-xl border border-red-200 bg-red-50 p-4 text-sm text-red-700">
|
<div className="rounded-xl border border-red-200 bg-red-50 p-4 text-sm text-red-700">
|
||||||
@@ -204,7 +356,20 @@ export function LiveRecorder({
|
|||||||
{hasTranscript ? (
|
{hasTranscript ? (
|
||||||
<div className="space-y-1 text-sm font-mono leading-relaxed text-neutral-700">
|
<div className="space-y-1 text-sm font-mono leading-relaxed text-neutral-700">
|
||||||
{chunks.map((chunk, i) => (
|
{chunks.map((chunk, i) => (
|
||||||
<p key={i}>{chunk.text}</p>
|
<p key={i}>
|
||||||
|
{chunk.speaker && (
|
||||||
|
<span
|
||||||
|
className={`mr-1.5 font-semibold ${
|
||||||
|
chunk.speaker === LOCAL_SPEAKER
|
||||||
|
? 'text-purple-600'
|
||||||
|
: 'text-teal-600'
|
||||||
|
}`}
|
||||||
|
>
|
||||||
|
{resolveSpeakerName(chunk.speaker, speakerNames)}:
|
||||||
|
</span>
|
||||||
|
)}
|
||||||
|
{chunk.text}
|
||||||
|
</p>
|
||||||
))}
|
))}
|
||||||
{interimText && (
|
{interimText && (
|
||||||
<p className="text-neutral-400 italic">
|
<p className="text-neutral-400 italic">
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
'use client'
|
||||||
|
|
||||||
|
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||||
|
|
||||||
|
export type CaptureMode = 'idle' | 'mic-only' | 'dual-stream'
|
||||||
|
|
||||||
|
interface DualStreamCapture {
|
||||||
|
mode: CaptureMode
|
||||||
|
/** 내 목소리 트랙 (마이크) */
|
||||||
|
micTrack: MediaStreamTrack | null
|
||||||
|
/** 상대 목소리 트랙 (시스템·탭 오디오). 화면 공유를 거부하거나 오디오를 안 켜면 null */
|
||||||
|
systemTrack: MediaStreamTrack | null
|
||||||
|
error: string | null
|
||||||
|
/** 시스템 오디오 없이 마이크만 잡힌 경우의 안내 */
|
||||||
|
warning: string | null
|
||||||
|
start: (options?: { withSystemAudio?: boolean }) => Promise<void>
|
||||||
|
stop: () => void
|
||||||
|
}
|
||||||
|
|
||||||
|
function describeCaptureError(err: unknown): string {
|
||||||
|
if (!(err instanceof Error)) return '오디오 캡처에 실패했습니다.'
|
||||||
|
|
||||||
|
switch (err.name) {
|
||||||
|
case 'NotAllowedError':
|
||||||
|
return '오디오 캡처 권한이 거부되었습니다. 마이크와 화면 공유를 모두 허용해주세요.'
|
||||||
|
case 'NotFoundError':
|
||||||
|
return '사용할 수 있는 마이크를 찾을 수 없습니다.'
|
||||||
|
case 'NotReadableError':
|
||||||
|
return '다른 앱이 마이크를 사용 중입니다. 해당 앱을 종료하고 다시 시도해주세요.'
|
||||||
|
default:
|
||||||
|
return `오디오 캡처 실패: ${err.message}`
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 화자 분리를 위한 2트랙 캡처.
|
||||||
|
*
|
||||||
|
* 내 마이크와 시스템(탭) 오디오를 **별도 트랙으로** 잡는다. 트랙이 곧 화자이므로
|
||||||
|
* 추론 없이 정확도 100%로 화자가 갈린다. 원격 1:1 회의를 겨냥한 경로다.
|
||||||
|
*
|
||||||
|
* 시스템 오디오는 Chrome/Windows와 Chrome 141+/macOS 14.2+ 에서만 캡처된다.
|
||||||
|
* 실패하면 마이크 단독 모드로 떨어지며, 그 경우 화자 분리는 되지 않는다.
|
||||||
|
*/
|
||||||
|
export function useDualStreamCapture(): DualStreamCapture {
|
||||||
|
const [mode, setMode] = useState<CaptureMode>('idle')
|
||||||
|
const [micTrack, setMicTrack] = useState<MediaStreamTrack | null>(null)
|
||||||
|
const [systemTrack, setSystemTrack] = useState<MediaStreamTrack | null>(null)
|
||||||
|
const [error, setError] = useState<string | null>(null)
|
||||||
|
const [warning, setWarning] = useState<string | null>(null)
|
||||||
|
|
||||||
|
const streamsRef = useRef<MediaStream[]>([])
|
||||||
|
|
||||||
|
const stop = useCallback(() => {
|
||||||
|
for (const stream of streamsRef.current) {
|
||||||
|
for (const track of stream.getTracks()) track.stop()
|
||||||
|
}
|
||||||
|
streamsRef.current = []
|
||||||
|
setMicTrack(null)
|
||||||
|
setSystemTrack(null)
|
||||||
|
setMode('idle')
|
||||||
|
}, [])
|
||||||
|
|
||||||
|
const start = useCallback(
|
||||||
|
async (options?: { withSystemAudio?: boolean }) => {
|
||||||
|
const withSystemAudio = options?.withSystemAudio ?? true
|
||||||
|
|
||||||
|
setError(null)
|
||||||
|
setWarning(null)
|
||||||
|
|
||||||
|
let micStream: MediaStream
|
||||||
|
try {
|
||||||
|
micStream = await navigator.mediaDevices.getUserMedia({ audio: true })
|
||||||
|
} catch (err) {
|
||||||
|
setError(describeCaptureError(err))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
streamsRef.current.push(micStream)
|
||||||
|
const mic = micStream.getAudioTracks()[0] ?? null
|
||||||
|
setMicTrack(mic)
|
||||||
|
|
||||||
|
if (!withSystemAudio) {
|
||||||
|
setMode('mic-only')
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
// 오디오만 필요해도 video:true를 함께 요청해야 한다 (스펙 요구사항).
|
||||||
|
const displayStream = await navigator.mediaDevices.getDisplayMedia({
|
||||||
|
video: true,
|
||||||
|
audio: true,
|
||||||
|
})
|
||||||
|
streamsRef.current.push(displayStream)
|
||||||
|
|
||||||
|
// 영상은 쓰지 않으므로 즉시 끊어 자원과 프라이버시 노출을 줄인다.
|
||||||
|
for (const track of displayStream.getVideoTracks()) track.stop()
|
||||||
|
|
||||||
|
const system = displayStream.getAudioTracks()[0] ?? null
|
||||||
|
if (!system) {
|
||||||
|
setWarning(
|
||||||
|
'시스템 오디오가 공유되지 않아 화자 분리가 비활성화됩니다. 공유 대화상자에서 "탭 오디오 공유"를 켜주세요.',
|
||||||
|
)
|
||||||
|
setMode('mic-only')
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
setSystemTrack(system)
|
||||||
|
setMode('dual-stream')
|
||||||
|
} catch (err) {
|
||||||
|
setWarning(
|
||||||
|
`${describeCaptureError(err)} 마이크만으로 계속 녹음합니다 (화자 분리 없음).`,
|
||||||
|
)
|
||||||
|
setMode('mic-only')
|
||||||
|
}
|
||||||
|
},
|
||||||
|
[],
|
||||||
|
)
|
||||||
|
|
||||||
|
useEffect(() => stop, [stop])
|
||||||
|
|
||||||
|
return { mode, micTrack, systemTrack, error, warning, start, stop }
|
||||||
|
}
|
||||||
@@ -3,6 +3,22 @@
|
|||||||
import { useState, useRef, useCallback, useEffect } from 'react'
|
import { useState, useRef, useCallback, useEffect } from 'react'
|
||||||
import type { TranscriptChunk } from '@/lib/transcript-formatter'
|
import type { TranscriptChunk } from '@/lib/transcript-formatter'
|
||||||
|
|
||||||
|
export interface UseSpeechRecognitionOptions {
|
||||||
|
/**
|
||||||
|
* 전사할 오디오 트랙. 지정하면 기본 마이크 대신 이 트랙을 인식한다.
|
||||||
|
* 트랙마다 인식기를 붙이면 화자가 트랙으로 확정된다.
|
||||||
|
*/
|
||||||
|
audioTrack?: MediaStreamTrack | null
|
||||||
|
/** 이 인식기가 만든 청크에 붙일 화자 식별자. */
|
||||||
|
speaker?: string
|
||||||
|
/**
|
||||||
|
* 타임스탬프 기준 시각(ms). 여러 인식기를 병합하려면 같은 값을 넘겨야
|
||||||
|
* 한 시간축 위에 놓인다. 생략하면 startListening 호출 시각을 쓴다.
|
||||||
|
*/
|
||||||
|
timeOrigin?: number
|
||||||
|
lang?: string
|
||||||
|
}
|
||||||
|
|
||||||
interface SpeechRecognitionHook {
|
interface SpeechRecognitionHook {
|
||||||
isListening: boolean
|
isListening: boolean
|
||||||
isSupported: boolean | null
|
isSupported: boolean | null
|
||||||
@@ -30,7 +46,10 @@ function describeError(code: string): string {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
export function useSpeechRecognition(): SpeechRecognitionHook {
|
export function useSpeechRecognition(
|
||||||
|
options: UseSpeechRecognitionOptions = {},
|
||||||
|
): SpeechRecognitionHook {
|
||||||
|
const { audioTrack, speaker, timeOrigin, lang = 'ko-KR' } = options
|
||||||
const [isListening, setIsListening] = useState(false)
|
const [isListening, setIsListening] = useState(false)
|
||||||
const [chunks, setChunks] = useState<TranscriptChunk[]>([])
|
const [chunks, setChunks] = useState<TranscriptChunk[]>([])
|
||||||
const [interimText, setInterimText] = useState('')
|
const [interimText, setInterimText] = useState('')
|
||||||
@@ -40,6 +59,12 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
|
|||||||
const startTimeRef = useRef<number>(0)
|
const startTimeRef = useRef<number>(0)
|
||||||
const networkRetryRef = useRef<number>(0)
|
const networkRetryRef = useRef<number>(0)
|
||||||
|
|
||||||
|
// start()/onend 콜백이 항상 최신 값을 보도록 ref로 들고 있는다.
|
||||||
|
const configRef = useRef({ audioTrack, speaker, timeOrigin, lang })
|
||||||
|
useEffect(() => {
|
||||||
|
configRef.current = { audioTrack, speaker, timeOrigin, lang }
|
||||||
|
}, [audioTrack, speaker, timeOrigin, lang])
|
||||||
|
|
||||||
const MAX_NETWORK_RETRIES = 8
|
const MAX_NETWORK_RETRIES = 8
|
||||||
|
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
@@ -59,12 +84,15 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
|
|||||||
const SpeechRecognitionAPI =
|
const SpeechRecognitionAPI =
|
||||||
window.SpeechRecognition || window.webkitSpeechRecognition
|
window.SpeechRecognition || window.webkitSpeechRecognition
|
||||||
|
|
||||||
|
const { audioTrack: track, speaker: speakerId, timeOrigin: origin, lang: language } =
|
||||||
|
configRef.current
|
||||||
|
|
||||||
const recognition = new SpeechRecognitionAPI()
|
const recognition = new SpeechRecognitionAPI()
|
||||||
recognition.lang = 'ko-KR'
|
recognition.lang = language
|
||||||
recognition.continuous = true
|
recognition.continuous = true
|
||||||
recognition.interimResults = true
|
recognition.interimResults = true
|
||||||
|
|
||||||
startTimeRef.current = Date.now()
|
startTimeRef.current = origin ?? Date.now()
|
||||||
networkRetryRef.current = 0
|
networkRetryRef.current = 0
|
||||||
setError(null)
|
setError(null)
|
||||||
|
|
||||||
@@ -88,6 +116,7 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
|
|||||||
startTime: Math.max(0, now - 2),
|
startTime: Math.max(0, now - 2),
|
||||||
endTime: now,
|
endTime: now,
|
||||||
isFinal: true,
|
isFinal: true,
|
||||||
|
...(speakerId ? { speaker: speakerId } : {}),
|
||||||
},
|
},
|
||||||
])
|
])
|
||||||
setInterimText('')
|
setInterimText('')
|
||||||
@@ -103,8 +132,14 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
|
|||||||
const delay = networkRetryRef.current > 0 ? 1500 : 0
|
const delay = networkRetryRef.current > 0 ? 1500 : 0
|
||||||
setTimeout(() => {
|
setTimeout(() => {
|
||||||
if (recognitionRef.current !== recognition) return
|
if (recognitionRef.current !== recognition) return
|
||||||
|
// 트랙이 끝났으면(화면 공유 중단 등) 재시작하지 않는다.
|
||||||
|
if (track && track.readyState !== 'live') {
|
||||||
|
setIsListening(false)
|
||||||
|
recognitionRef.current = null
|
||||||
|
return
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
recognition.start()
|
recognition.start(track ?? undefined)
|
||||||
} catch {
|
} catch {
|
||||||
setIsListening(false)
|
setIsListening(false)
|
||||||
recognitionRef.current = null
|
recognitionRef.current = null
|
||||||
@@ -141,7 +176,7 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
|
|||||||
|
|
||||||
recognitionRef.current = recognition
|
recognitionRef.current = recognition
|
||||||
try {
|
try {
|
||||||
recognition.start()
|
recognition.start(track ?? undefined)
|
||||||
setIsListening(true)
|
setIsListening(true)
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
setError(
|
setError(
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
/**
|
||||||
|
* 화자 식별자.
|
||||||
|
*
|
||||||
|
* dual-stream 캡처에서는 트랙이 곧 화자이므로 추론이 개입하지 않는다.
|
||||||
|
* 단일 트랙에서 화자를 추정하는 경로가 생기면 'spk_1' 같은 식별자가 추가된다.
|
||||||
|
*/
|
||||||
|
export const LOCAL_SPEAKER = 'local'
|
||||||
|
export const REMOTE_SPEAKER = 'remote'
|
||||||
|
|
||||||
|
export type SpeakerNames = Readonly<Record<string, string>>
|
||||||
|
|
||||||
|
const DEFAULT_NAMES: SpeakerNames = {
|
||||||
|
[LOCAL_SPEAKER]: '나',
|
||||||
|
[REMOTE_SPEAKER]: '상대',
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 화자 식별자를 표시용 이름으로 바꾼다. 사용자가 지정한 이름이 우선한다. */
|
||||||
|
export function resolveSpeakerName(
|
||||||
|
speaker: string,
|
||||||
|
names?: SpeakerNames,
|
||||||
|
): string {
|
||||||
|
const custom = names?.[speaker]?.trim()
|
||||||
|
if (custom) return custom
|
||||||
|
|
||||||
|
const preset = DEFAULT_NAMES[speaker]
|
||||||
|
if (preset) return preset
|
||||||
|
|
||||||
|
return speaker
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 회의록에 저장할 참석자 목록. 지정된 이름이 없으면 기본 표기를 쓴다. */
|
||||||
|
export function listAttendees(
|
||||||
|
speakers: readonly string[],
|
||||||
|
names?: SpeakerNames,
|
||||||
|
): string[] {
|
||||||
|
const seen = new Set<string>()
|
||||||
|
const result: string[] = []
|
||||||
|
|
||||||
|
for (const speaker of speakers) {
|
||||||
|
if (seen.has(speaker)) continue
|
||||||
|
seen.add(speaker)
|
||||||
|
result.push(resolveSpeakerName(speaker, names))
|
||||||
|
}
|
||||||
|
|
||||||
|
return result
|
||||||
|
}
|
||||||
@@ -1,8 +1,12 @@
|
|||||||
|
import { resolveSpeakerName, type SpeakerNames } from './speakers'
|
||||||
|
|
||||||
export interface TranscriptChunk {
|
export interface TranscriptChunk {
|
||||||
text: string
|
text: string
|
||||||
startTime: number
|
startTime: number
|
||||||
endTime: number
|
endTime: number
|
||||||
isFinal: boolean
|
isFinal: boolean
|
||||||
|
/** 화자 식별자. 단일 트랙 녹음에서는 없다. */
|
||||||
|
speaker?: string
|
||||||
}
|
}
|
||||||
|
|
||||||
function formatTimestamp(seconds: number): string {
|
function formatTimestamp(seconds: number): string {
|
||||||
@@ -16,14 +20,42 @@ function formatTimestamp(seconds: number): string {
|
|||||||
return `[${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
|
return `[${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
|
||||||
}
|
}
|
||||||
|
|
||||||
export function formatTranscriptChunks(chunks: readonly TranscriptChunk[]): string {
|
export function formatTranscriptChunks(
|
||||||
|
chunks: readonly TranscriptChunk[],
|
||||||
|
names?: SpeakerNames,
|
||||||
|
): string {
|
||||||
if (chunks.length === 0) return ''
|
if (chunks.length === 0) return ''
|
||||||
|
|
||||||
return chunks
|
return chunks
|
||||||
.map((chunk) => `${formatTimestamp(chunk.startTime)} ${chunk.text}`)
|
.map((chunk) => {
|
||||||
|
const stamp = formatTimestamp(chunk.startTime)
|
||||||
|
if (!chunk.speaker) return `${stamp} ${chunk.text}`
|
||||||
|
return `${stamp} ${resolveSpeakerName(chunk.speaker, names)}: ${chunk.text}`
|
||||||
|
})
|
||||||
.join('\n')
|
.join('\n')
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 여러 트랙에서 나온 청크를 하나의 시간축으로 합친다.
|
||||||
|
*
|
||||||
|
* 각 인식기가 독립적으로 동작하므로 도착 순서가 발화 순서와 다를 수 있다.
|
||||||
|
* 모든 트랙이 같은 timeOrigin을 공유한다는 전제 아래 startTime으로 정렬한다.
|
||||||
|
*/
|
||||||
|
export function mergeSpeakerChunks(
|
||||||
|
...streams: readonly (readonly TranscriptChunk[])[]
|
||||||
|
): TranscriptChunk[] {
|
||||||
|
const merged = streams.flat()
|
||||||
|
|
||||||
|
return merged
|
||||||
|
.map((chunk, index) => ({ chunk, index }))
|
||||||
|
.sort((a, b) => {
|
||||||
|
const diff = a.chunk.startTime - b.chunk.startTime
|
||||||
|
// 같은 시각이면 원래 순서를 유지해 결과가 흔들리지 않게 한다.
|
||||||
|
return diff !== 0 ? diff : a.index - b.index
|
||||||
|
})
|
||||||
|
.map(({ chunk }) => chunk)
|
||||||
|
}
|
||||||
|
|
||||||
export function mergeAdjacentChunks(
|
export function mergeAdjacentChunks(
|
||||||
chunks: readonly TranscriptChunk[],
|
chunks: readonly TranscriptChunk[],
|
||||||
gapThreshold: number,
|
gapThreshold: number,
|
||||||
@@ -36,12 +68,19 @@ export function mergeAdjacentChunks(
|
|||||||
const current = chunks[i]
|
const current = chunks[i]
|
||||||
const last = result[result.length - 1]
|
const last = result[result.length - 1]
|
||||||
|
|
||||||
|
// 화자가 다르면 붙이지 않는다 — 서로 다른 사람의 발화가 한 덩어리가 되면 안 된다.
|
||||||
|
if (current.speaker !== last.speaker) {
|
||||||
|
result.push({ ...current })
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
if (current.startTime - last.endTime <= gapThreshold) {
|
if (current.startTime - last.endTime <= gapThreshold) {
|
||||||
result[result.length - 1] = {
|
result[result.length - 1] = {
|
||||||
text: `${last.text} ${current.text}`,
|
text: `${last.text} ${current.text}`,
|
||||||
startTime: last.startTime,
|
startTime: last.startTime,
|
||||||
endTime: current.endTime,
|
endTime: current.endTime,
|
||||||
isFinal: current.isFinal,
|
isFinal: current.isFinal,
|
||||||
|
...(last.speaker ? { speaker: last.speaker } : {}),
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
result.push({ ...current })
|
result.push({ ...current })
|
||||||
|
|||||||
Vendored
+5
-1
@@ -5,7 +5,11 @@ interface SpeechRecognition extends EventTarget {
|
|||||||
onresult: ((event: SpeechRecognitionEvent) => void) | null
|
onresult: ((event: SpeechRecognitionEvent) => void) | null
|
||||||
onend: (() => void) | null
|
onend: (() => void) | null
|
||||||
onerror: ((event: SpeechRecognitionErrorEvent) => void) | null
|
onerror: ((event: SpeechRecognitionErrorEvent) => void) | null
|
||||||
start(): void
|
/**
|
||||||
|
* audioTrack을 넘기면 기본 마이크 대신 그 트랙을 전사한다 (Chrome 한정).
|
||||||
|
* 트랙별로 인식기를 붙이면 화자가 트랙 번호로 확정된다.
|
||||||
|
*/
|
||||||
|
start(audioTrack?: MediaStreamTrack): void
|
||||||
stop(): void
|
stop(): void
|
||||||
abort(): void
|
abort(): void
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in new issue
Block a user