mirror of
https://github.com/nad4-su/meeting-minutes.git
synced 2026-10-11 15:50:13 +09:00
perf: 롤링 요약을 증분 방식으로 전환해 토큰 비용을 선형화 (#15)
롤링 요약이 30초마다 누적 전사 전체를 다시 보내고 있었다. 호출 N회차가 그때까지의 전사 전부를 담으므로 총 토큰이 회의 길이의 제곱으로 늘어난다. 1시간 회의 기준 약 116만 입력 토큰, 2시간이면 2배가 아니라 4배가 된다. [지금까지의 요약] + [새로 추가된 발화]만 보내도록 바꿔 호출당 토큰을 회의 길이와 무관하게 일정하게 만들었다. - templates.ts: BuildPromptArgs.incremental 추가, INCREMENTAL_MODIFIER 신설. 증분 모드에서는 전사 블록이 자체 라벨을 갖는다 - live-summary.ts: previousSummary 옵션. 값이 있으면 증분 모드로 동작 - live-summary.ts: planLiveSummaryRequest() — 전체/증분 결정을 순수 함수로 분리해 단위 테스트 가능하게 함 - useLiveSummary: 마지막 요약 지점 인덱스를 추적해 델타만 전송. 성공했을 때만 전진시켜 실패해도 발화를 잃지 않는다 - 요약을 요약하는 구조라 오차가 누적되므로 fullRefreshEvery(기본 20회)마다 전사 전체로 한 번 다시 요약해 오차를 끊는다 기본 설정에서 1시간 회의 기준 약 7배, 전체 재요약을 끄면 약 12배 절감된다. 회의가 2~3분보다 짧으면 지시문 오버헤드 때문에 오히려 조금 늘어난다. 테스트 141 → 151. Co-authored-by: csbae <csbae@RP-002.local> Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
5 files changed
+276
-8
No files matched your search
@@ -1,5 +1,8 @@
|
||||
import { describe, it, expect, vi } from 'vitest'
|
||||
import { generateLiveSummary } from '@/lib/live-summary'
|
||||
import {
|
||||
generateLiveSummary,
|
||||
planLiveSummaryRequest,
|
||||
} from '@/lib/live-summary'
|
||||
|
||||
describe('generateLiveSummary', () => {
|
||||
it('Gemini API 응답을 받아 중간 요약 마크다운을 반환한다', async () => {
|
||||
@@ -155,3 +158,131 @@ describe('generateLiveSummary — openai-compatible', () => {
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
describe('planLiveSummaryRequest', () => {
|
||||
const base = {
|
||||
totalChunks: 50,
|
||||
lastSummarizedIndex: 30,
|
||||
incrementsSinceFull: 3,
|
||||
fullRefreshEvery: 20,
|
||||
hasPreviousSummary: true,
|
||||
}
|
||||
|
||||
it('평상시에는 직전 지점부터 증분으로 보낸다', () => {
|
||||
expect(planLiveSummaryRequest(base)).toEqual({
|
||||
mode: 'incremental',
|
||||
startIndex: 30,
|
||||
})
|
||||
})
|
||||
|
||||
it('첫 호출은 전체를 보낸다', () => {
|
||||
expect(
|
||||
planLiveSummaryRequest({ ...base, lastSummarizedIndex: 0 }),
|
||||
).toEqual({ mode: 'full', startIndex: 0 })
|
||||
})
|
||||
|
||||
it('갱신할 직전 요약이 없으면 전체를 보낸다', () => {
|
||||
expect(
|
||||
planLiveSummaryRequest({ ...base, hasPreviousSummary: false }),
|
||||
).toEqual({ mode: 'full', startIndex: 0 })
|
||||
})
|
||||
|
||||
it('증분이 누적되면 전체 재요약으로 오차를 끊는다', () => {
|
||||
expect(
|
||||
planLiveSummaryRequest({ ...base, incrementsSinceFull: 20 }),
|
||||
).toEqual({ mode: 'full', startIndex: 0 })
|
||||
})
|
||||
|
||||
it('fullRefreshEvery=0이면 전체 재요약을 하지 않는다', () => {
|
||||
expect(
|
||||
planLiveSummaryRequest({
|
||||
...base,
|
||||
fullRefreshEvery: 0,
|
||||
incrementsSinceFull: 999,
|
||||
}).mode,
|
||||
).toBe('incremental')
|
||||
})
|
||||
|
||||
it('전사가 초기화되어 인덱스가 범위를 벗어나면 전체를 보낸다', () => {
|
||||
expect(
|
||||
planLiveSummaryRequest({ ...base, totalChunks: 5 }),
|
||||
).toEqual({ mode: 'full', startIndex: 0 })
|
||||
})
|
||||
})
|
||||
|
||||
describe('generateLiveSummary — 증분 모드', () => {
|
||||
function mockOk(text = '## 요약\n갱신됨') {
|
||||
return vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: () =>
|
||||
Promise.resolve({
|
||||
candidates: [{ content: { parts: [{ text }] } }],
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
function sentPrompt(mockFetch: ReturnType<typeof vi.fn>): string {
|
||||
const body = JSON.parse(mockFetch.mock.calls[0][1].body)
|
||||
return body.contents[0].parts[0].text
|
||||
}
|
||||
|
||||
it('previousSummary가 있으면 요약과 신규 발화를 나눠 전달한다', async () => {
|
||||
const mockFetch = mockOk()
|
||||
|
||||
await generateLiveSummary('새로 나온 이야기', {
|
||||
apiKey: 'k',
|
||||
previousSummary: '## 요약\n이전까지의 내용',
|
||||
fetchFn: mockFetch,
|
||||
})
|
||||
|
||||
const prompt = sentPrompt(mockFetch)
|
||||
expect(prompt).toContain('[지금까지의 요약]')
|
||||
expect(prompt).toContain('이전까지의 내용')
|
||||
expect(prompt).toContain('[새로 추가된 발화]')
|
||||
expect(prompt).toContain('새로 나온 이야기')
|
||||
expect(prompt).toContain('갱신')
|
||||
})
|
||||
|
||||
it('previousSummary가 없으면 기존 전체 요약 형식을 유지한다', async () => {
|
||||
const mockFetch = mockOk()
|
||||
|
||||
await generateLiveSummary('전체 전사', { apiKey: 'k', fetchFn: mockFetch })
|
||||
|
||||
const prompt = sentPrompt(mockFetch)
|
||||
expect(prompt).toContain('음성 인식 텍스트:')
|
||||
expect(prompt).not.toContain('[지금까지의 요약]')
|
||||
})
|
||||
|
||||
it('빈 문자열 previousSummary는 증분으로 취급하지 않는다', async () => {
|
||||
const mockFetch = mockOk()
|
||||
|
||||
await generateLiveSummary('전체 전사', {
|
||||
apiKey: 'k',
|
||||
previousSummary: ' ',
|
||||
fetchFn: mockFetch,
|
||||
})
|
||||
|
||||
expect(sentPrompt(mockFetch)).not.toContain('[지금까지의 요약]')
|
||||
})
|
||||
|
||||
it('긴 회의에서 증분 프롬프트가 전체 프롬프트보다 짧다', async () => {
|
||||
const longTranscript = '회의 발화 한 줄입니다.\n'.repeat(500)
|
||||
|
||||
const fullFetch = mockOk()
|
||||
await generateLiveSummary(longTranscript, {
|
||||
apiKey: 'k',
|
||||
fetchFn: fullFetch,
|
||||
})
|
||||
|
||||
const incFetch = mockOk()
|
||||
await generateLiveSummary('마지막 30초에 나온 이야기', {
|
||||
apiKey: 'k',
|
||||
previousSummary: '## 요약\n지금까지의 요약 본문',
|
||||
fetchFn: incFetch,
|
||||
})
|
||||
|
||||
expect(sentPrompt(incFetch).length).toBeLessThan(
|
||||
sentPrompt(fullFetch).length / 5,
|
||||
)
|
||||
})
|
||||
})
|
||||
@@ -4,11 +4,12 @@ import { resolveProviderSettings } from '@/lib/api-keys'
|
||||
import type { SummaryDepth, TemplateId } from '@/lib/templates'
|
||||
|
||||
const MAX_TRANSCRIPT_CHARS = 40_000
|
||||
const MAX_PREVIOUS_SUMMARY_CHARS = 8_000
|
||||
|
||||
export async function POST(request: NextRequest) {
|
||||
try {
|
||||
const body = await request.json()
|
||||
const { transcript, template, depth, customPrompt } = body
|
||||
const { transcript, template, depth, customPrompt, previousSummary } = body
|
||||
|
||||
if (!transcript || typeof transcript !== 'string') {
|
||||
return Response.json(
|
||||
@@ -33,8 +34,15 @@ export async function POST(request: NextRequest) {
|
||||
? transcript.slice(-MAX_TRANSCRIPT_CHARS)
|
||||
: transcript
|
||||
|
||||
// 증분 모드: 직전 요약 + 새 발화만 보내므로 호출당 토큰이 일정하다.
|
||||
const previous =
|
||||
typeof previousSummary === 'string'
|
||||
? previousSummary.slice(-MAX_PREVIOUS_SUMMARY_CHARS)
|
||||
: undefined
|
||||
|
||||
const result = await generateLiveSummary(truncated, {
|
||||
provider: settings,
|
||||
previousSummary: previous,
|
||||
template: (template as TemplateId | undefined) ?? 'meeting',
|
||||
depth: depth as SummaryDepth | undefined,
|
||||
customPrompt:
|
||||
|
||||
@@ -5,12 +5,15 @@ import type { TranscriptChunk } from '@/lib/transcript-formatter'
|
||||
import { formatTranscriptChunks } from '@/lib/transcript-formatter'
|
||||
import type { SummaryDepth, TemplateId } from '@/lib/templates'
|
||||
import { getProviderRequestPayload } from '@/lib/api-key-storage'
|
||||
import { planLiveSummaryRequest } from '@/lib/live-summary'
|
||||
|
||||
interface UseLiveSummaryOptions {
|
||||
enabled: boolean
|
||||
pollIntervalMs?: number
|
||||
minWords?: number
|
||||
incrementWords?: number
|
||||
/** 증분 요약을 이 횟수만큼 반복하면 전사 전체로 한 번 다시 요약한다. 0이면 끈다. */
|
||||
fullRefreshEvery?: number
|
||||
template?: TemplateId
|
||||
depth?: SummaryDepth
|
||||
customPrompt?: string
|
||||
@@ -41,6 +44,7 @@ export function useLiveSummary(
|
||||
pollIntervalMs = 30_000,
|
||||
minWords = 25,
|
||||
incrementWords = 40,
|
||||
fullRefreshEvery = 20,
|
||||
template = 'meeting',
|
||||
depth,
|
||||
customPrompt,
|
||||
@@ -64,6 +68,11 @@ export function useLiveSummary(
|
||||
const cooldownUntilRef = useRef<number | null>(null)
|
||||
const consecutiveFailuresRef = useRef(0)
|
||||
|
||||
// 증분 요약 상태 — 성공했을 때만 전진시켜서 실패해도 발화를 잃지 않는다.
|
||||
const summaryRef = useRef('')
|
||||
const lastSummarizedIndexRef = useRef(0)
|
||||
const incrementsSinceFullRef = useRef(0)
|
||||
|
||||
useEffect(() => {
|
||||
chunksRef.current = chunks
|
||||
}, [chunks])
|
||||
@@ -98,12 +107,31 @@ export function useLiveSummary(
|
||||
return
|
||||
}
|
||||
|
||||
const transcript = formatTranscriptChunks(chunksRef.current)
|
||||
const allChunks = chunksRef.current
|
||||
const transcript = formatTranscriptChunks(allChunks)
|
||||
const currentWordCount = countWords(transcript)
|
||||
|
||||
if (currentWordCount < minWords) return
|
||||
if (currentWordCount - lastWordCountRef.current < incrementWords) return
|
||||
|
||||
const plan = planLiveSummaryRequest({
|
||||
totalChunks: allChunks.length,
|
||||
lastSummarizedIndex: lastSummarizedIndexRef.current,
|
||||
incrementsSinceFull: incrementsSinceFullRef.current,
|
||||
fullRefreshEvery,
|
||||
hasPreviousSummary: summaryRef.current.trim().length > 0,
|
||||
})
|
||||
|
||||
// 요청을 보내는 시점의 길이를 고정해 둔다. 응답을 기다리는 동안
|
||||
// 새 청크가 쌓여도 그 부분은 다음 회차로 넘어간다.
|
||||
const chunkCountAtSend = allChunks.length
|
||||
const payloadTranscript =
|
||||
plan.mode === 'full'
|
||||
? transcript
|
||||
: formatTranscriptChunks(allChunks.slice(plan.startIndex))
|
||||
|
||||
if (payloadTranscript.trim().length === 0) return
|
||||
|
||||
const controller = new AbortController()
|
||||
abortRef.current = controller
|
||||
inFlightRef.current = true
|
||||
@@ -115,7 +143,9 @@ export function useLiveSummary(
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
transcript,
|
||||
transcript: payloadTranscript,
|
||||
previousSummary:
|
||||
plan.mode === 'incremental' ? summaryRef.current : undefined,
|
||||
template: configRef.current.template,
|
||||
depth: configRef.current.depth,
|
||||
customPrompt: configRef.current.customPrompt,
|
||||
@@ -142,8 +172,12 @@ export function useLiveSummary(
|
||||
|
||||
const data = await res.json()
|
||||
setSummary(data.markdown)
|
||||
summaryRef.current = data.markdown
|
||||
setLastUpdatedAt(Date.now())
|
||||
lastWordCountRef.current = currentWordCount
|
||||
lastSummarizedIndexRef.current = chunkCountAtSend
|
||||
incrementsSinceFullRef.current =
|
||||
plan.mode === 'full' ? 0 : incrementsSinceFullRef.current + 1
|
||||
consecutiveFailuresRef.current = 0
|
||||
cooldownUntilRef.current = null
|
||||
setCooldownUntil(null)
|
||||
@@ -164,7 +198,7 @@ export function useLiveSummary(
|
||||
return () => {
|
||||
clearInterval(interval)
|
||||
}
|
||||
}, [enabled, pollIntervalMs, minWords, incrementWords])
|
||||
}, [enabled, pollIntervalMs, minWords, incrementWords, fullRefreshEvery])
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
|
||||
+66
-1
@@ -14,6 +14,14 @@ interface LiveSummaryOptions {
|
||||
/** 프로바이더 설정. 생략하면 apiKey로 Gemini를 호출한다. */
|
||||
provider?: ProviderSettings
|
||||
apiKey?: string
|
||||
/**
|
||||
* 직전 롤링 요약.
|
||||
*
|
||||
* 값이 있으면 증분 모드로 동작하며, 이때 `transcript` 인자는 전사 전체가 아니라
|
||||
* **직전 요약 이후 새로 추가된 발화**만 담아야 한다. 호출당 토큰이 회의 길이와
|
||||
* 무관하게 일정해진다.
|
||||
*/
|
||||
previousSummary?: string
|
||||
template?: TemplateId
|
||||
depth?: SummaryDepth
|
||||
customPrompt?: string
|
||||
@@ -32,13 +40,19 @@ export async function generateLiveSummary(
|
||||
return { success: false, error: '요약할 텍스트가 비어 있습니다.' }
|
||||
}
|
||||
|
||||
const previous = options.previousSummary?.trim() ?? ''
|
||||
const incremental = previous.length > 0
|
||||
|
||||
const templateId = options.template ?? 'meeting'
|
||||
const depth = resolveDepth(templateId, options.depth)
|
||||
const prompt = buildPrompt({
|
||||
templateId,
|
||||
depth,
|
||||
transcript: trimmed,
|
||||
transcript: incremental
|
||||
? `[지금까지의 요약]\n${previous}\n\n[새로 추가된 발화]\n${trimmed}`
|
||||
: trimmed,
|
||||
live: true,
|
||||
incremental,
|
||||
customPrompt: options.customPrompt,
|
||||
})
|
||||
|
||||
@@ -58,3 +72,54 @@ export async function generateLiveSummary(
|
||||
|
||||
return { success: true, markdown: result.text }
|
||||
}
|
||||
|
||||
export interface LiveSummaryPlan {
|
||||
/** 'full'이면 전사 전체를 보내고 previousSummary를 쓰지 않는다. */
|
||||
mode: 'full' | 'incremental'
|
||||
/** 이번에 보낼 청크의 시작 인덱스. full이면 0. */
|
||||
startIndex: number
|
||||
}
|
||||
|
||||
export interface LiveSummaryPlanArgs {
|
||||
totalChunks: number
|
||||
lastSummarizedIndex: number
|
||||
incrementsSinceFull: number
|
||||
/** 0이면 주기적 전체 재요약을 하지 않는다. */
|
||||
fullRefreshEvery: number
|
||||
hasPreviousSummary: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* 이번 롤링 요약 호출을 증분으로 보낼지 전체로 보낼지 결정한다.
|
||||
*
|
||||
* 증분 모드는 호출당 토큰을 회의 길이와 무관하게 유지하지만, 요약을 요약하는
|
||||
* 구조라 반복될수록 오차가 쌓인다. 그래서 일정 횟수마다 전사 전체로 한 번씩
|
||||
* 다시 요약해 오차를 끊는다.
|
||||
*/
|
||||
export function planLiveSummaryRequest(
|
||||
args: LiveSummaryPlanArgs,
|
||||
): LiveSummaryPlan {
|
||||
const {
|
||||
totalChunks,
|
||||
lastSummarizedIndex,
|
||||
incrementsSinceFull,
|
||||
fullRefreshEvery,
|
||||
hasPreviousSummary,
|
||||
} = args
|
||||
|
||||
// 전사가 초기화되어 인덱스가 범위를 벗어난 경우
|
||||
if (lastSummarizedIndex > totalChunks) {
|
||||
return { mode: 'full', startIndex: 0 }
|
||||
}
|
||||
|
||||
// 첫 호출이거나 갱신할 요약이 아직 없는 경우
|
||||
if (lastSummarizedIndex === 0 || !hasPreviousSummary) {
|
||||
return { mode: 'full', startIndex: 0 }
|
||||
}
|
||||
|
||||
if (fullRefreshEvery > 0 && incrementsSinceFull >= fullRefreshEvery) {
|
||||
return { mode: 'full', startIndex: 0 }
|
||||
}
|
||||
|
||||
return { mode: 'incremental', startIndex: lastSummarizedIndex }
|
||||
}
|
||||
+32
-2
@@ -197,11 +197,27 @@ const DEPTH_MODIFIERS: Record<SummaryDepth, string> = {
|
||||
|
||||
const LIVE_MODIFIER = `회의가 아직 진행 중입니다. 지금까지의 내용을 기반으로 **중간 정리**를 작성하세요. 확정되지 않은 결정은 "(논의 중)"으로 표시하세요.`
|
||||
|
||||
/**
|
||||
* 증분 갱신 모드.
|
||||
*
|
||||
* 전사 전체를 매번 다시 보내는 대신 [지금까지의 요약] + [새로 추가된 발화]만 보낸다.
|
||||
* 호출당 토큰이 회의 길이와 무관하게 일정해져, 비용이 제곱이 아닌 선형으로 늘어난다.
|
||||
*/
|
||||
const INCREMENTAL_MODIFIER = `아래에는 [지금까지의 요약]과 [새로 추가된 발화]가 주어집니다.
|
||||
기존 요약을 처음부터 다시 쓰지 말고 **갱신**하세요:
|
||||
- 기존 요약의 내용과 구조를 유지한 채 새 발화를 반영합니다.
|
||||
- 기존 항목이 새 발화로 확정되거나 번복되었다면 그 항목을 고치세요.
|
||||
- 새 발화에 언급되지 않았다는 이유로 기존 내용을 삭제하지 마세요.
|
||||
- 출력은 항상 갱신된 회의록 **전체**입니다. 변경분만 출력하지 마세요.`
|
||||
|
||||
export interface BuildPromptArgs {
|
||||
templateId: TemplateId
|
||||
depth: SummaryDepth
|
||||
/** 증분 모드에서는 [지금까지의 요약] + [새로 추가된 발화]를 담은 블록이 들어온다. */
|
||||
transcript: string
|
||||
live?: boolean
|
||||
/** 직전 요약을 갱신하는 모드. live와 함께 쓴다. */
|
||||
incremental?: boolean
|
||||
customPrompt?: string
|
||||
}
|
||||
|
||||
@@ -222,7 +238,14 @@ export function resolveDepth(
|
||||
}
|
||||
|
||||
export function buildPrompt(args: BuildPromptArgs): string {
|
||||
const { templateId, depth, transcript, live = false, customPrompt } = args
|
||||
const {
|
||||
templateId,
|
||||
depth,
|
||||
transcript,
|
||||
live = false,
|
||||
incremental = false,
|
||||
customPrompt,
|
||||
} = args
|
||||
|
||||
let instruction: string
|
||||
|
||||
@@ -243,12 +266,19 @@ export function buildPrompt(args: BuildPromptArgs): string {
|
||||
|
||||
if (live && template.liveSupported) {
|
||||
parts.push(LIVE_MODIFIER)
|
||||
|
||||
if (incremental) {
|
||||
parts.push(INCREMENTAL_MODIFIER)
|
||||
}
|
||||
}
|
||||
|
||||
instruction = parts.join('\n\n')
|
||||
}
|
||||
|
||||
return `${instruction}\n\n---\n음성 인식 텍스트:\n${transcript}`
|
||||
// 증분 모드의 transcript는 자체 라벨([지금까지의 요약] 등)을 이미 포함한다.
|
||||
const body = incremental ? transcript : `음성 인식 텍스트:\n${transcript}`
|
||||
|
||||
return `${instruction}\n\n---\n${body}`
|
||||
}
|
||||
|
||||
export function liveEnabledFor(templateId: TemplateId): boolean {
|
||||
|
||||
Reference in new issue
Block a user