Author SHA1 Message Date
csbaeandClaude Opus 5 a8cfab8a4a fix: 두 트랙을 확보했을 때만 화자를 라벨링
대면 회의처럼 시스템 오디오가 없는 상황에서 화자 분리를 켜면,
마이크에 섞여 들어온 상대 발언까지 "나"로 기록되는 문제가 있었다.
라벨이 틀리는 것은 라벨이 없는 것보다 나쁘다.

capture.mode가 'dual-stream'일 때만 speaker를 붙이고, 마이크 단독으로
폴백하면 라벨 없이 기존 형식으로 기록한다.

설정 UI에도 대면 회의에서는 동작하지 않는다는 점을 명시했다.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-07 18:29:58 +09:00
csbaeandClaude Opus 5 d2cb5b0009 feat: dual-stream 캡처로 원격 회의 화자 분리
내 마이크와 시스템(탭) 오디오를 별도 트랙으로 잡고, 트랙마다 인식기를
붙인다. 트랙이 곧 화자이므로 추론도 모델 다운로드도 없이 화자가 갈린다.

Chrome이 SpeechRecognition.start(audioTrack)를 지원하는 덕분에 가능하다.
트랙을 명시적으로 넘기므로 기본 마이크를 두고 경합할 일도 없다.

- speakers.ts 신설 — 화자 식별자와 표시 이름 해석
- TranscriptChunk.speaker 추가, formatTranscriptChunks가 "화자: 발화"로 출력
- mergeSpeakerChunks — 트랙별 청크를 하나의 시간축으로 병합.
  여러 인식기가 같은 timeOrigin을 공유해야 순서가 맞는다
- mergeAdjacentChunks가 화자 경계를 넘어 합치지 않도록 수정
- useSpeechRecognition: audioTrack / speaker / timeOrigin / lang 옵션.
  인자 없이 호출하면 기존 동작 그대로다
- useDualStreamCapture 신설 — getUserMedia + getDisplayMedia.
  video는 즉시 끊고 오디오만 쓴다. 시스템 오디오 확보에 실패하면
  마이크 단독으로 폴백하고 화자 분리가 꺼졌음을 안내한다
- LiveRecorder: 화자 분리 토글, 화자 이름 입력, 전사 목록 화자 색상 구분

원격 다자 회의와 대면 회의는 원격 트랙에 diarization이 필요하며 후속 작업이다.

테스트 162 → 176.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-07 18:27:00 +09:00
8 changed files with 555 additions and 19 deletions

No files matched your search

+45
View File
@@ -0,0 +1,45 @@
import { describe, it, expect } from 'vitest'
import {
LOCAL_SPEAKER,
REMOTE_SPEAKER,
listAttendees,
resolveSpeakerName,
} from '@/lib/speakers'
describe('resolveSpeakerName', () => {
it('기본 화자에 한국어 표기를 붙인다', () => {
expect(resolveSpeakerName(LOCAL_SPEAKER)).toBe('나')
expect(resolveSpeakerName(REMOTE_SPEAKER)).toBe('상대')
})
it('사용자가 지정한 이름이 우선한다', () => {
const names = { [REMOTE_SPEAKER]: '김지훈' }
expect(resolveSpeakerName(REMOTE_SPEAKER, names)).toBe('김지훈')
expect(resolveSpeakerName(LOCAL_SPEAKER, names)).toBe('나')
})
it('공백뿐인 이름은 무시한다', () => {
expect(resolveSpeakerName(LOCAL_SPEAKER, { [LOCAL_SPEAKER]: ' ' })).toBe('나')
})
it('모르는 식별자는 그대로 돌려준다', () => {
expect(resolveSpeakerName('spk_3')).toBe('spk_3')
})
})
describe('listAttendees', () => {
it('중복을 제거하고 순서를 유지한다', () => {
expect(
listAttendees([REMOTE_SPEAKER, LOCAL_SPEAKER, REMOTE_SPEAKER]),
).toEqual(['상대', '나'])
})
it('지정된 이름을 반영한다', () => {
expect(
listAttendees([LOCAL_SPEAKER, REMOTE_SPEAKER], {
[LOCAL_SPEAKER]: '배철승',
[REMOTE_SPEAKER]: '김지훈',
}),
).toEqual(['배철승', '김지훈'])
})
})
@@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest'
import {
formatTranscriptChunks,
mergeAdjacentChunks,
mergeSpeakerChunks,
type TranscriptChunk,
} from '@/lib/transcript-formatter'
@@ -55,3 +56,83 @@ describe('mergeAdjacentChunks', () => {
expect(result).toEqual([])
})
})
describe('화자 표기', () => {
it('speaker가 있으면 이름을 붙여 출력한다', () => {
const out = formatTranscriptChunks([
{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'local' },
{ text: '네 반갑습니다', startTime: 3, endTime: 5, isFinal: true, speaker: 'remote' },
])
expect(out).toBe('[00:00] 나: 안녕하세요\n[00:03] 상대: 네 반갑습니다')
})
it('지정된 화자 이름을 사용한다', () => {
const out = formatTranscriptChunks(
[{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true, speaker: 'remote' }],
{ remote: '김지훈' },
)
expect(out).toBe('[00:00] 김지훈: 안녕하세요')
})
it('speaker가 없으면 기존 형식을 유지한다', () => {
const out = formatTranscriptChunks([
{ text: '안녕하세요', startTime: 0, endTime: 2, isFinal: true },
])
expect(out).toBe('[00:00] 안녕하세요')
})
})
describe('mergeSpeakerChunks', () => {
const local = [
{ text: '먼저 말함', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' },
{ text: '나중에 말함', startTime: 9, endTime: 10, isFinal: true, speaker: 'local' },
]
const remote = [
{ text: '중간에 답함', startTime: 4, endTime: 6, isFinal: true, speaker: 'remote' },
]
it('여러 트랙을 시간순으로 합친다', () => {
expect(mergeSpeakerChunks(local, remote).map((c) => c.text)).toEqual([
'먼저 말함',
'중간에 답함',
'나중에 말함',
])
})
it('같은 시각이면 넘긴 순서를 유지한다', () => {
const a = [{ text: 'A', startTime: 5, endTime: 6, isFinal: true, speaker: 'local' }]
const b = [{ text: 'B', startTime: 5, endTime: 6, isFinal: true, speaker: 'remote' }]
expect(mergeSpeakerChunks(a, b).map((c) => c.text)).toEqual(['A', 'B'])
})
it('빈 트랙을 섞어도 동작한다', () => {
expect(mergeSpeakerChunks([], remote, [])).toEqual(remote)
expect(mergeSpeakerChunks()).toEqual([])
})
})
describe('mergeAdjacentChunks — 화자 경계', () => {
it('화자가 다르면 간격이 좁아도 합치지 않는다', () => {
const merged = mergeAdjacentChunks(
[
{ text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' },
{ text: '어렵습니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'remote' },
],
5,
)
expect(merged).toHaveLength(2)
})
it('같은 화자면 기존대로 합친다', () => {
const merged = mergeAdjacentChunks(
[
{ text: '그건', startTime: 0, endTime: 1, isFinal: true, speaker: 'local' },
{ text: '가능합니다', startTime: 1, endTime: 2, isFinal: true, speaker: 'local' },
],
5,
)
expect(merged).toHaveLength(1)
expect(merged[0].text).toBe('그건 가능합니다')
expect(merged[0].speaker).toBe('local')
})
})
+176 -11
View File
@@ -1,9 +1,18 @@
'use client'
import { useEffect, useRef } from 'react'
import { useEffect, useMemo, useRef, useState } from 'react'
import { useSpeechRecognition } from '@/hooks/useSpeechRecognition'
import { useDualStreamCapture } from '@/hooks/useDualStreamCapture'
import { useLiveSummary } from '@/hooks/useLiveSummary'
import { formatTranscriptChunks } from '@/lib/transcript-formatter'
import {
formatTranscriptChunks,
mergeSpeakerChunks,
} from '@/lib/transcript-formatter'
import {
LOCAL_SPEAKER,
REMOTE_SPEAKER,
resolveSpeakerName,
} from '@/lib/speakers'
import type { SummaryDepth, TemplateId } from '@/lib/templates'
interface LiveRecorderProps {
@@ -21,16 +30,57 @@ export function LiveRecorder({
depth,
customPrompt,
}: LiveRecorderProps) {
const [speakerSeparation, setSpeakerSeparation] = useState(false)
const [pendingStart, setPendingStart] = useState(false)
const [speakerNames, setSpeakerNames] = useState<Record<string, string>>({})
const timeOriginRef = useRef(0)
const capture = useDualStreamCapture()
// 화자 분리를 켜면 트랙마다 인식기를 붙인다. 트랙이 곧 화자가 되므로
// 추론 없이 화자가 갈린다.
// 두 트랙을 실제로 확보했을 때만 화자를 라벨링한다.
// 마이크만 잡힌 상태에서 라벨을 붙이면 상대 발언이 내 것으로 기록된다.
const isSeparating = capture.mode === 'dual-stream'
const local = useSpeechRecognition({
audioTrack: speakerSeparation ? capture.micTrack : null,
speaker: isSeparating ? LOCAL_SPEAKER : undefined,
timeOrigin: timeOriginRef.current || undefined,
})
const remote = useSpeechRecognition({
audioTrack: capture.systemTrack,
speaker: REMOTE_SPEAKER,
timeOrigin: timeOriginRef.current || undefined,
})
const {
isListening,
isSupported,
chunks,
interimText,
error: recognitionError,
startListening,
stopListening,
resetChunks,
} = useSpeechRecognition()
} = local
const chunks = useMemo(
() =>
isSeparating
? mergeSpeakerChunks(local.chunks, remote.chunks)
: local.chunks,
[isSeparating, local.chunks, remote.chunks],
)
// 캡처가 끝나 트랙이 준비되면 그때 인식기를 시작한다.
useEffect(() => {
if (!pendingStart) return
if (capture.mode === 'idle') return
setPendingStart(false)
local.startListening()
if (capture.systemTrack) remote.startListening()
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [pendingStart, capture.mode, capture.systemTrack])
const {
summary,
@@ -78,14 +128,34 @@ export function LiveRecorder({
)
}
async function handleStart() {
timeOriginRef.current = Date.now()
if (!speakerSeparation) {
startListening()
return
}
setPendingStart(true)
await capture.start({ withSystemAudio: true })
}
function handleStop() {
stopListening()
const transcript = formatTranscriptChunks(chunks)
local.stopListening()
remote.stopListening()
capture.stop()
const transcript = formatTranscriptChunks(chunks, speakerNames)
if (transcript.length > 0) {
onTranscriptReady(transcript)
}
}
function handleReset() {
local.resetChunks()
remote.resetChunks()
}
const lastUpdatedLabel = lastUpdatedAt
? new Date(lastUpdatedAt).toLocaleTimeString('ko-KR', {
hour: '2-digit',
@@ -136,7 +206,7 @@ export function LiveRecorder({
</button>
) : (
<button
onClick={startListening}
onClick={handleStart}
className="flex items-center gap-2 rounded-full bg-blue-600 px-6 py-3 text-white font-medium shadow-lg shadow-blue-600/25 hover:bg-blue-700 transition-colors"
>
<span className="text-lg">🎤</span>
@@ -146,7 +216,7 @@ export function LiveRecorder({
{chunks.length > 0 && !isListening && (
<button
onClick={resetChunks}
onClick={handleReset}
className="rounded-full border border-neutral-300 px-4 py-2 text-sm text-neutral-600 hover:bg-neutral-50 transition-colors"
>
초기화
@@ -159,8 +229,90 @@ export function LiveRecorder({
녹음 중 · {wordCount}단어 · {chunks.length}개 구간
</div>
)}
{isListening && capture.mode === 'dual-stream' && (
<div className="flex items-center gap-1.5 rounded-full bg-purple-50 px-4 py-1.5 text-xs text-purple-700">
<span className="h-2 w-2 rounded-full bg-purple-500" />
화자 분리 중
</div>
)}
</div>
{!isListening && chunks.length === 0 && (
<div className="rounded-xl border border-neutral-200 bg-white p-4">
<label className="flex items-start gap-3 cursor-pointer">
<input
type="checkbox"
checked={speakerSeparation}
onChange={(e) => setSpeakerSeparation(e.target.checked)}
className="mt-0.5 h-4 w-4 accent-purple-600"
/>
<span>
<span className="text-sm font-medium text-neutral-800">
화자 분리 (원격 회의)
</span>
<span className="mt-0.5 block text-xs text-neutral-500">
내 마이크와 상대 목소리를 별도 트랙으로 받아 구분합니다. 추론이 없어
정확합니다. 시작 시 <strong>화면 공유 대화상자에서 &ldquo;탭 오디오 공유&rdquo;를
반드시 켜주세요.</strong>
</span>
<span className="mt-1.5 block text-xs text-amber-700">
한 대의 노트북을 앞에 두고 마주 앉은 <strong>대면 회의에서는 동작하지
않습니다.</strong> 마이크 하나에 두 사람 목소리가 섞여 들어오기 때문입니다.
이 경우 화자 라벨 없이 녹음되며, 대면 화자 분리는 준비 중입니다.
</span>
</span>
</label>
{speakerSeparation && (
<div className="mt-4 grid gap-3 sm:grid-cols-2">
<label className="block">
<span className="mb-1 block text-xs text-neutral-500">내 이름</span>
<input
type="text"
value={speakerNames[LOCAL_SPEAKER] ?? ''}
onChange={(e) =>
setSpeakerNames((prev) => ({
...prev,
[LOCAL_SPEAKER]: e.target.value,
}))
}
placeholder="나"
className="w-full rounded-lg border border-neutral-300 px-3 py-2 text-sm focus:border-purple-400 focus:outline-none"
/>
</label>
<label className="block">
<span className="mb-1 block text-xs text-neutral-500">상대 이름</span>
<input
type="text"
value={speakerNames[REMOTE_SPEAKER] ?? ''}
onChange={(e) =>
setSpeakerNames((prev) => ({
...prev,
[REMOTE_SPEAKER]: e.target.value,
}))
}
placeholder="상대"
className="w-full rounded-lg border border-neutral-300 px-3 py-2 text-sm focus:border-purple-400 focus:outline-none"
/>
</label>
</div>
)}
</div>
)}
{capture.error && (
<div className="rounded-xl border border-red-200 bg-red-50 p-4 text-sm text-red-700">
<strong>오디오 캡처 오류:</strong> {capture.error}
</div>
)}
{capture.warning && (
<div className="rounded-xl border border-amber-200 bg-amber-50 p-4 text-sm text-amber-800">
⚠️ {capture.warning}
</div>
)}
{recognitionError && (
<div className="rounded-xl border border-red-200 bg-red-50 p-4 text-sm text-red-700">
<strong>음성 인식 오류:</strong> {recognitionError}
@@ -204,7 +356,20 @@ export function LiveRecorder({
{hasTranscript ? (
<div className="space-y-1 text-sm font-mono leading-relaxed text-neutral-700">
{chunks.map((chunk, i) => (
<p key={i}>{chunk.text}</p>
<p key={i}>
{chunk.speaker && (
<span
className={`mr-1.5 font-semibold ${
chunk.speaker === LOCAL_SPEAKER
? 'text-purple-600'
: 'text-teal-600'
}`}
>
{resolveSpeakerName(chunk.speaker, speakerNames)}:
</span>
)}
{chunk.text}
</p>
))}
{interimText && (
<p className="text-neutral-400 italic">
+121
View File
@@ -0,0 +1,121 @@
'use client'
import { useCallback, useEffect, useRef, useState } from 'react'
export type CaptureMode = 'idle' | 'mic-only' | 'dual-stream'
interface DualStreamCapture {
mode: CaptureMode
/** 내 목소리 트랙 (마이크) */
micTrack: MediaStreamTrack | null
/** 상대 목소리 트랙 (시스템·탭 오디오). 화면 공유를 거부하거나 오디오를 안 켜면 null */
systemTrack: MediaStreamTrack | null
error: string | null
/** 시스템 오디오 없이 마이크만 잡힌 경우의 안내 */
warning: string | null
start: (options?: { withSystemAudio?: boolean }) => Promise<void>
stop: () => void
}
function describeCaptureError(err: unknown): string {
if (!(err instanceof Error)) return '오디오 캡처에 실패했습니다.'
switch (err.name) {
case 'NotAllowedError':
return '오디오 캡처 권한이 거부되었습니다. 마이크와 화면 공유를 모두 허용해주세요.'
case 'NotFoundError':
return '사용할 수 있는 마이크를 찾을 수 없습니다.'
case 'NotReadableError':
return '다른 앱이 마이크를 사용 중입니다. 해당 앱을 종료하고 다시 시도해주세요.'
default:
return `오디오 캡처 실패: ${err.message}`
}
}
/**
* 화자 분리를 위한 2트랙 캡처.
*
* 내 마이크와 시스템(탭) 오디오를 **별도 트랙으로** 잡는다. 트랙이 곧 화자이므로
* 추론 없이 정확도 100%로 화자가 갈린다. 원격 1:1 회의를 겨냥한 경로다.
*
* 시스템 오디오는 Chrome/Windows와 Chrome 141+/macOS 14.2+ 에서만 캡처된다.
* 실패하면 마이크 단독 모드로 떨어지며, 그 경우 화자 분리는 되지 않는다.
*/
export function useDualStreamCapture(): DualStreamCapture {
const [mode, setMode] = useState<CaptureMode>('idle')
const [micTrack, setMicTrack] = useState<MediaStreamTrack | null>(null)
const [systemTrack, setSystemTrack] = useState<MediaStreamTrack | null>(null)
const [error, setError] = useState<string | null>(null)
const [warning, setWarning] = useState<string | null>(null)
const streamsRef = useRef<MediaStream[]>([])
const stop = useCallback(() => {
for (const stream of streamsRef.current) {
for (const track of stream.getTracks()) track.stop()
}
streamsRef.current = []
setMicTrack(null)
setSystemTrack(null)
setMode('idle')
}, [])
const start = useCallback(
async (options?: { withSystemAudio?: boolean }) => {
const withSystemAudio = options?.withSystemAudio ?? true
setError(null)
setWarning(null)
let micStream: MediaStream
try {
micStream = await navigator.mediaDevices.getUserMedia({ audio: true })
} catch (err) {
setError(describeCaptureError(err))
return
}
streamsRef.current.push(micStream)
const mic = micStream.getAudioTracks()[0] ?? null
setMicTrack(mic)
if (!withSystemAudio) {
setMode('mic-only')
return
}
try {
// 오디오만 필요해도 video:true를 함께 요청해야 한다 (스펙 요구사항).
const displayStream = await navigator.mediaDevices.getDisplayMedia({
video: true,
audio: true,
})
streamsRef.current.push(displayStream)
// 영상은 쓰지 않으므로 즉시 끊어 자원과 프라이버시 노출을 줄인다.
for (const track of displayStream.getVideoTracks()) track.stop()
const system = displayStream.getAudioTracks()[0] ?? null
if (!system) {
setWarning(
'시스템 오디오가 공유되지 않아 화자 분리가 비활성화됩니다. 공유 대화상자에서 "탭 오디오 공유"를 켜주세요.',
)
setMode('mic-only')
return
}
setSystemTrack(system)
setMode('dual-stream')
} catch (err) {
setWarning(
`${describeCaptureError(err)} 마이크만으로 계속 녹음합니다 (화자 분리 없음).`,
)
setMode('mic-only')
}
},
[],
)
useEffect(() => stop, [stop])
return { mode, micTrack, systemTrack, error, warning, start, stop }
}
+40 -5
View File
@@ -3,6 +3,22 @@
import { useState, useRef, useCallback, useEffect } from 'react'
import type { TranscriptChunk } from '@/lib/transcript-formatter'
export interface UseSpeechRecognitionOptions {
/**
* 전사할 오디오 트랙. 지정하면 기본 마이크 대신 이 트랙을 인식한다.
* 트랙마다 인식기를 붙이면 화자가 트랙으로 확정된다.
*/
audioTrack?: MediaStreamTrack | null
/** 이 인식기가 만든 청크에 붙일 화자 식별자. */
speaker?: string
/**
* 타임스탬프 기준 시각(ms). 여러 인식기를 병합하려면 같은 값을 넘겨야
* 한 시간축 위에 놓인다. 생략하면 startListening 호출 시각을 쓴다.
*/
timeOrigin?: number
lang?: string
}
interface SpeechRecognitionHook {
isListening: boolean
isSupported: boolean | null
@@ -30,7 +46,10 @@ function describeError(code: string): string {
}
}
export function useSpeechRecognition(): SpeechRecognitionHook {
export function useSpeechRecognition(
options: UseSpeechRecognitionOptions = {},
): SpeechRecognitionHook {
const { audioTrack, speaker, timeOrigin, lang = 'ko-KR' } = options
const [isListening, setIsListening] = useState(false)
const [chunks, setChunks] = useState<TranscriptChunk[]>([])
const [interimText, setInterimText] = useState('')
@@ -40,6 +59,12 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
const startTimeRef = useRef<number>(0)
const networkRetryRef = useRef<number>(0)
// start()/onend 콜백이 항상 최신 값을 보도록 ref로 들고 있는다.
const configRef = useRef({ audioTrack, speaker, timeOrigin, lang })
useEffect(() => {
configRef.current = { audioTrack, speaker, timeOrigin, lang }
}, [audioTrack, speaker, timeOrigin, lang])
const MAX_NETWORK_RETRIES = 8
useEffect(() => {
@@ -59,12 +84,15 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
const SpeechRecognitionAPI =
window.SpeechRecognition || window.webkitSpeechRecognition
const { audioTrack: track, speaker: speakerId, timeOrigin: origin, lang: language } =
configRef.current
const recognition = new SpeechRecognitionAPI()
recognition.lang = 'ko-KR'
recognition.lang = language
recognition.continuous = true
recognition.interimResults = true
startTimeRef.current = Date.now()
startTimeRef.current = origin ?? Date.now()
networkRetryRef.current = 0
setError(null)
@@ -88,6 +116,7 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
startTime: Math.max(0, now - 2),
endTime: now,
isFinal: true,
...(speakerId ? { speaker: speakerId } : {}),
},
])
setInterimText('')
@@ -103,8 +132,14 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
const delay = networkRetryRef.current > 0 ? 1500 : 0
setTimeout(() => {
if (recognitionRef.current !== recognition) return
// 트랙이 끝났으면(화면 공유 중단 등) 재시작하지 않는다.
if (track && track.readyState !== 'live') {
setIsListening(false)
recognitionRef.current = null
return
}
try {
recognition.start()
recognition.start(track ?? undefined)
} catch {
setIsListening(false)
recognitionRef.current = null
@@ -141,7 +176,7 @@ export function useSpeechRecognition(): SpeechRecognitionHook {
recognitionRef.current = recognition
try {
recognition.start()
recognition.start(track ?? undefined)
setIsListening(true)
} catch (err) {
setError(
+46
View File
@@ -0,0 +1,46 @@
/**
* 화자 식별자.
*
* dual-stream 캡처에서는 트랙이 곧 화자이므로 추론이 개입하지 않는다.
* 단일 트랙에서 화자를 추정하는 경로가 생기면 'spk_1' 같은 식별자가 추가된다.
*/
export const LOCAL_SPEAKER = 'local'
export const REMOTE_SPEAKER = 'remote'
export type SpeakerNames = Readonly<Record<string, string>>
const DEFAULT_NAMES: SpeakerNames = {
[LOCAL_SPEAKER]: '나',
[REMOTE_SPEAKER]: '상대',
}
/** 화자 식별자를 표시용 이름으로 바꾼다. 사용자가 지정한 이름이 우선한다. */
export function resolveSpeakerName(
speaker: string,
names?: SpeakerNames,
): string {
const custom = names?.[speaker]?.trim()
if (custom) return custom
const preset = DEFAULT_NAMES[speaker]
if (preset) return preset
return speaker
}
/** 회의록에 저장할 참석자 목록. 지정된 이름이 없으면 기본 표기를 쓴다. */
export function listAttendees(
speakers: readonly string[],
names?: SpeakerNames,
): string[] {
const seen = new Set<string>()
const result: string[] = []
for (const speaker of speakers) {
if (seen.has(speaker)) continue
seen.add(speaker)
result.push(resolveSpeakerName(speaker, names))
}
return result
}
+41 -2
View File
@@ -1,8 +1,12 @@
import { resolveSpeakerName, type SpeakerNames } from './speakers'
export interface TranscriptChunk {
text: string
startTime: number
endTime: number
isFinal: boolean
/** 화자 식별자. 단일 트랙 녹음에서는 없다. */
speaker?: string
}
function formatTimestamp(seconds: number): string {
@@ -16,14 +20,42 @@ function formatTimestamp(seconds: number): string {
return `[${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
}
export function formatTranscriptChunks(chunks: readonly TranscriptChunk[]): string {
export function formatTranscriptChunks(
chunks: readonly TranscriptChunk[],
names?: SpeakerNames,
): string {
if (chunks.length === 0) return ''
return chunks
.map((chunk) => `${formatTimestamp(chunk.startTime)} ${chunk.text}`)
.map((chunk) => {
const stamp = formatTimestamp(chunk.startTime)
if (!chunk.speaker) return `${stamp} ${chunk.text}`
return `${stamp} ${resolveSpeakerName(chunk.speaker, names)}: ${chunk.text}`
})
.join('\n')
}
/**
* 여러 트랙에서 나온 청크를 하나의 시간축으로 합친다.
*
* 각 인식기가 독립적으로 동작하므로 도착 순서가 발화 순서와 다를 수 있다.
* 모든 트랙이 같은 timeOrigin을 공유한다는 전제 아래 startTime으로 정렬한다.
*/
export function mergeSpeakerChunks(
...streams: readonly (readonly TranscriptChunk[])[]
): TranscriptChunk[] {
const merged = streams.flat()
return merged
.map((chunk, index) => ({ chunk, index }))
.sort((a, b) => {
const diff = a.chunk.startTime - b.chunk.startTime
// 같은 시각이면 원래 순서를 유지해 결과가 흔들리지 않게 한다.
return diff !== 0 ? diff : a.index - b.index
})
.map(({ chunk }) => chunk)
}
export function mergeAdjacentChunks(
chunks: readonly TranscriptChunk[],
gapThreshold: number,
@@ -36,12 +68,19 @@ export function mergeAdjacentChunks(
const current = chunks[i]
const last = result[result.length - 1]
// 화자가 다르면 붙이지 않는다 — 서로 다른 사람의 발화가 한 덩어리가 되면 안 된다.
if (current.speaker !== last.speaker) {
result.push({ ...current })
continue
}
if (current.startTime - last.endTime <= gapThreshold) {
result[result.length - 1] = {
text: `${last.text} ${current.text}`,
startTime: last.startTime,
endTime: current.endTime,
isFinal: current.isFinal,
...(last.speaker ? { speaker: last.speaker } : {}),
}
} else {
result.push({ ...current })
+5 -1
View File
@@ -5,7 +5,11 @@ interface SpeechRecognition extends EventTarget {
onresult: ((event: SpeechRecognitionEvent) => void) | null
onend: (() => void) | null
onerror: ((event: SpeechRecognitionErrorEvent) => void) | null
start(): void
/**
* audioTrack을 넘기면 기본 마이크 대신 그 트랙을 전사한다 (Chrome 한정).
* 트랙별로 인식기를 붙이면 화자가 트랙 번호로 확정된다.
*/
start(audioTrack?: MediaStreamTrack): void
stop(): void
abort(): void
}