mirror of
https://github.com/nad4-su/meeting-minutes.git
synced 2026-10-11 15:50:13 +09:00
마이크 한 트랙에 여러 사람이 섞여 들어오는 대면 회의에서 화자를 나눈다. pyannote-segmentation-3.0 + 3D-Speaker CAM++ 를 ONNX Runtime으로 로컬 Next.js 프로세스에서 실행한다. GPU도 파이썬도 필요 없고 오디오가 외부로 나가지 않는다. 실측 (Apple Silicon, 2스레드, 4인 한국어 아닌 샘플 57초): - 처리 1.5초 → RTF 0.03, 실시간의 33배 - 자동 추정은 화자 5명(과다 계수), 참석자 수를 주면 4명(정확) → 참석자 수 입력을 1급 기능으로 노출 구성: - lib/diarization.ts — sherpa-onnx-node 래퍼, 모델 캐시, 참석자 수 힌트 - lib/audio-session.ts — 녹음 중 PCM 조각을 세션별로 누적 (메모리) - api/diarize/chunk — 4초짜리 Int16 PCM 조각 업로드 - api/diarize — 누적 오디오 전체 분석, GET은 모델 설치 여부 확인 - public/pcm-worklet.js — 16kHz Int16 PCM 추출 AudioWorklet - hooks/useDiarization.ts — 수집·분석 상태 관리 - transcript-formatter: assignSpeakers() — 시간 겹침으로 화자 배정. 겹치는 구간이 없으면 라벨을 비운다 (틀린 라벨보다 없는 편이 낫다) - LiveRecorder: 사용 안 함 / 원격 회의 / 대면 회의 3모드 + 참석자 수 - scripts/setup-diarization.mjs — 모델 34MB 내려받기 (npm run setup:diarization) 설계 변경: 롤링 재-diarization 대신 녹음 종료 시 1회 전체 분석. 구간을 잘라 반복하면 실행마다 화자 번호가 달라져 이어 붙일 수 없고, 누적분 재분석은 비용이 회의 길이에 비례해 커진다. 대신 실시간 화자 표시는 없다. 모델은 라이선스와 용량 때문에 저장소에 넣지 않고 gitignore 한다. 테스트 176 → 184. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
153 lines
4.2 KiB
TypeScript
153 lines
4.2 KiB
TypeScript
import { resolveSpeakerName, type SpeakerNames } from './speakers'
|
|
|
|
export interface TranscriptChunk {
|
|
text: string
|
|
startTime: number
|
|
endTime: number
|
|
isFinal: boolean
|
|
/** 화자 식별자. 단일 트랙 녹음에서는 없다. */
|
|
speaker?: string
|
|
}
|
|
|
|
function formatTimestamp(seconds: number): string {
|
|
const h = Math.floor(seconds / 3600)
|
|
const m = Math.floor((seconds % 3600) / 60)
|
|
const s = Math.floor(seconds % 60)
|
|
|
|
if (h > 0) {
|
|
return `[${String(h).padStart(2, '0')}:${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
|
|
}
|
|
return `[${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}]`
|
|
}
|
|
|
|
export function formatTranscriptChunks(
|
|
chunks: readonly TranscriptChunk[],
|
|
names?: SpeakerNames,
|
|
): string {
|
|
if (chunks.length === 0) return ''
|
|
|
|
return chunks
|
|
.map((chunk) => {
|
|
const stamp = formatTimestamp(chunk.startTime)
|
|
if (!chunk.speaker) return `${stamp} ${chunk.text}`
|
|
return `${stamp} ${resolveSpeakerName(chunk.speaker, names)}: ${chunk.text}`
|
|
})
|
|
.join('\n')
|
|
}
|
|
|
|
/**
|
|
* 여러 트랙에서 나온 청크를 하나의 시간축으로 합친다.
|
|
*
|
|
* 각 인식기가 독립적으로 동작하므로 도착 순서가 발화 순서와 다를 수 있다.
|
|
* 모든 트랙이 같은 timeOrigin을 공유한다는 전제 아래 startTime으로 정렬한다.
|
|
*/
|
|
export function mergeSpeakerChunks(
|
|
...streams: readonly (readonly TranscriptChunk[])[]
|
|
): TranscriptChunk[] {
|
|
const merged = streams.flat()
|
|
|
|
return merged
|
|
.map((chunk, index) => ({ chunk, index }))
|
|
.sort((a, b) => {
|
|
const diff = a.chunk.startTime - b.chunk.startTime
|
|
// 같은 시각이면 원래 순서를 유지해 결과가 흔들리지 않게 한다.
|
|
return diff !== 0 ? diff : a.index - b.index
|
|
})
|
|
.map(({ chunk }) => chunk)
|
|
}
|
|
|
|
export function mergeAdjacentChunks(
|
|
chunks: readonly TranscriptChunk[],
|
|
gapThreshold: number,
|
|
): TranscriptChunk[] {
|
|
if (chunks.length === 0) return []
|
|
|
|
const result: TranscriptChunk[] = [{ ...chunks[0] }]
|
|
|
|
for (let i = 1; i < chunks.length; i++) {
|
|
const current = chunks[i]
|
|
const last = result[result.length - 1]
|
|
|
|
// 화자가 다르면 붙이지 않는다 — 서로 다른 사람의 발화가 한 덩어리가 되면 안 된다.
|
|
if (current.speaker !== last.speaker) {
|
|
result.push({ ...current })
|
|
continue
|
|
}
|
|
|
|
if (current.startTime - last.endTime <= gapThreshold) {
|
|
result[result.length - 1] = {
|
|
text: `${last.text} ${current.text}`,
|
|
startTime: last.startTime,
|
|
endTime: current.endTime,
|
|
isFinal: current.isFinal,
|
|
...(last.speaker ? { speaker: last.speaker } : {}),
|
|
}
|
|
} else {
|
|
result.push({ ...current })
|
|
}
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
export interface TimeSpan {
|
|
start: number
|
|
end: number
|
|
speaker: number
|
|
}
|
|
|
|
function overlap(
|
|
aStart: number,
|
|
aEnd: number,
|
|
bStart: number,
|
|
bEnd: number,
|
|
): number {
|
|
return Math.max(0, Math.min(aEnd, bEnd) - Math.max(aStart, bStart))
|
|
}
|
|
|
|
/**
|
|
* diarization 결과를 전사 청크에 붙인다.
|
|
*
|
|
* 청크마다 시간이 가장 많이 겹치는 화자 구간을 고른다. 겹치는 구간이 없으면
|
|
* 화자를 비워 둔다 — 틀린 화자를 붙이는 것보다 없는 편이 낫다.
|
|
*
|
|
* Web Speech API가 주는 청크 타임스탬프는 근사값이므로 이 매칭도 근사다.
|
|
*/
|
|
export function assignSpeakers(
|
|
chunks: readonly TranscriptChunk[],
|
|
segments: readonly TimeSpan[],
|
|
speakerId: (index: number) => string,
|
|
): TranscriptChunk[] {
|
|
if (segments.length === 0) return chunks.map((chunk) => ({ ...chunk }))
|
|
|
|
return chunks.map((chunk) => {
|
|
let best: TimeSpan | null = null
|
|
let bestOverlap = 0
|
|
|
|
for (const segment of segments) {
|
|
const value = overlap(
|
|
chunk.startTime,
|
|
chunk.endTime,
|
|
segment.start,
|
|
segment.end,
|
|
)
|
|
if (value > bestOverlap) {
|
|
bestOverlap = value
|
|
best = segment
|
|
}
|
|
}
|
|
|
|
if (!best) {
|
|
// 겹치는 구간이 없으면 이전 라벨도 남기지 않는다.
|
|
return {
|
|
text: chunk.text,
|
|
startTime: chunk.startTime,
|
|
endTime: chunk.endTime,
|
|
isFinal: chunk.isFinal,
|
|
}
|
|
}
|
|
|
|
return { ...chunk, speaker: speakerId(best.speaker) }
|
|
})
|
|
}
|