mirror of
https://github.com/nad4-su/meeting-minutes.git
synced 2026-10-11 15:50:13 +09:00
마이크 한 트랙에 여러 사람이 섞여 들어오는 대면 회의에서 화자를 나눈다. pyannote-segmentation-3.0 + 3D-Speaker CAM++ 를 ONNX Runtime으로 로컬 Next.js 프로세스에서 실행한다. GPU도 파이썬도 필요 없고 오디오가 외부로 나가지 않는다. 실측 (Apple Silicon, 2스레드, 4인 한국어 아닌 샘플 57초): - 처리 1.5초 → RTF 0.03, 실시간의 33배 - 자동 추정은 화자 5명(과다 계수), 참석자 수를 주면 4명(정확) → 참석자 수 입력을 1급 기능으로 노출 구성: - lib/diarization.ts — sherpa-onnx-node 래퍼, 모델 캐시, 참석자 수 힌트 - lib/audio-session.ts — 녹음 중 PCM 조각을 세션별로 누적 (메모리) - api/diarize/chunk — 4초짜리 Int16 PCM 조각 업로드 - api/diarize — 누적 오디오 전체 분석, GET은 모델 설치 여부 확인 - public/pcm-worklet.js — 16kHz Int16 PCM 추출 AudioWorklet - hooks/useDiarization.ts — 수집·분석 상태 관리 - transcript-formatter: assignSpeakers() — 시간 겹침으로 화자 배정. 겹치는 구간이 없으면 라벨을 비운다 (틀린 라벨보다 없는 편이 낫다) - LiveRecorder: 사용 안 함 / 원격 회의 / 대면 회의 3모드 + 참석자 수 - scripts/setup-diarization.mjs — 모델 34MB 내려받기 (npm run setup:diarization) 설계 변경: 롤링 재-diarization 대신 녹음 종료 시 1회 전체 분석. 구간을 잘라 반복하면 실행마다 화자 번호가 달라져 이어 붙일 수 없고, 누적분 재분석은 비용이 회의 길이에 비례해 커진다. 대신 실시간 화자 표시는 없다. 모델은 라이선스와 용량 때문에 저장소에 넣지 않고 gitignore 한다. 테스트 176 → 184. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
99 lines
2.9 KiB
JavaScript
99 lines
2.9 KiB
JavaScript
#!/usr/bin/env node
|
|
/**
|
|
* 화자 분리 모델 내려받기.
|
|
*
|
|
* 두 모델 모두 CPU에서 도는 ONNX 파일이며 GPU나 파이썬 런타임이 필요 없다.
|
|
* 라이선스 문제로 저장소에 포함하지 않고 최초 1회 내려받는다.
|
|
*
|
|
* npm run setup:diarization
|
|
*/
|
|
import { createWriteStream } from 'node:fs'
|
|
import { mkdir, rename, rm, stat } from 'node:fs/promises'
|
|
import { pipeline } from 'node:stream/promises'
|
|
import path from 'node:path'
|
|
import { spawn } from 'node:child_process'
|
|
|
|
const DEST = path.join(process.cwd(), 'models', 'diarization')
|
|
|
|
const MODELS = [
|
|
{
|
|
name: 'segmentation.onnx',
|
|
// pyannote/segmentation-3.0 (MIT) — 발화 구간 분할
|
|
url: 'https://github.com/k2-fsa/sherpa-onnx/releases/download/speaker-segmentation-models/sherpa-onnx-pyannote-segmentation-3-0.tar.bz2',
|
|
archiveEntry: 'sherpa-onnx-pyannote-segmentation-3-0/model.onnx',
|
|
approxMB: 6,
|
|
},
|
|
{
|
|
name: 'embedding.onnx',
|
|
// 3D-Speaker CAM++ — 화자 임베딩
|
|
url: 'https://github.com/k2-fsa/sherpa-onnx/releases/download/speaker-recongition-models/3dspeaker_speech_campplus_sv_zh-cn_16k-common.onnx',
|
|
approxMB: 28,
|
|
},
|
|
]
|
|
|
|
async function exists(p) {
|
|
try {
|
|
await stat(p)
|
|
return true
|
|
} catch {
|
|
return false
|
|
}
|
|
}
|
|
|
|
async function download(url, dest) {
|
|
const res = await fetch(url, { redirect: 'follow' })
|
|
if (!res.ok) throw new Error(`${res.status} ${res.statusText} — ${url}`)
|
|
await pipeline(res.body, createWriteStream(dest))
|
|
}
|
|
|
|
function run(cmd, args, cwd) {
|
|
return new Promise((resolve, reject) => {
|
|
const child = spawn(cmd, args, { cwd, stdio: 'inherit' })
|
|
child.on('error', reject)
|
|
child.on('exit', (code) =>
|
|
code === 0 ? resolve() : reject(new Error(`${cmd} 실패 (exit ${code})`)),
|
|
)
|
|
})
|
|
}
|
|
|
|
async function main() {
|
|
await mkdir(DEST, { recursive: true })
|
|
|
|
for (const model of MODELS) {
|
|
const target = path.join(DEST, model.name)
|
|
|
|
if (await exists(target)) {
|
|
console.log(`✓ ${model.name} — 이미 있음`)
|
|
continue
|
|
}
|
|
|
|
console.log(`↓ ${model.name} (약 ${model.approxMB}MB) 내려받는 중...`)
|
|
|
|
if (!model.archiveEntry) {
|
|
await download(model.url, target)
|
|
console.log(`✓ ${model.name}`)
|
|
continue
|
|
}
|
|
|
|
const archive = path.join(DEST, 'tmp.tar.bz2')
|
|
await download(model.url, archive)
|
|
await run('tar', ['xjf', archive], DEST)
|
|
await rename(path.join(DEST, model.archiveEntry), target)
|
|
await rm(archive)
|
|
await rm(path.join(DEST, path.dirname(model.archiveEntry)), {
|
|
recursive: true,
|
|
force: true,
|
|
})
|
|
console.log(`✓ ${model.name}`)
|
|
}
|
|
|
|
console.log(`\n완료. 모델 위치: ${DEST}`)
|
|
console.log('이제 녹음 화면에서 "대면 회의 화자 분리"를 쓸 수 있습니다.')
|
|
}
|
|
|
|
main().catch((err) => {
|
|
console.error(`\n✗ 실패: ${err.message}`)
|
|
console.error('네트워크를 확인하고 다시 실행해주세요.')
|
|
process.exit(1)
|
|
})
|