perf: 롤링 요약을 증분 방식으로 전환해 토큰 비용을 선형화 (#15)

롤링 요약이 30초마다 누적 전사 전체를 다시 보내고 있었다. 호출 N회차가
그때까지의 전사 전부를 담으므로 총 토큰이 회의 길이의 제곱으로 늘어난다.
1시간 회의 기준 약 116만 입력 토큰, 2시간이면 2배가 아니라 4배가 된다.

[지금까지의 요약] + [새로 추가된 발화]만 보내도록 바꿔 호출당 토큰을
회의 길이와 무관하게 일정하게 만들었다.

- templates.ts: BuildPromptArgs.incremental 추가, INCREMENTAL_MODIFIER 신설.
  증분 모드에서는 전사 블록이 자체 라벨을 갖는다
- live-summary.ts: previousSummary 옵션. 값이 있으면 증분 모드로 동작
- live-summary.ts: planLiveSummaryRequest() — 전체/증분 결정을 순수 함수로
  분리해 단위 테스트 가능하게 함
- useLiveSummary: 마지막 요약 지점 인덱스를 추적해 델타만 전송.
  성공했을 때만 전진시켜 실패해도 발화를 잃지 않는다
- 요약을 요약하는 구조라 오차가 누적되므로 fullRefreshEvery(기본 20회)마다
  전사 전체로 한 번 다시 요약해 오차를 끊는다

기본 설정에서 1시간 회의 기준 약 7배, 전체 재요약을 끄면 약 12배 절감된다.
회의가 2~3분보다 짧으면 지시문 오버헤드 때문에 오히려 조금 늘어난다.

테스트 141 → 151.

Co-authored-by: csbae <csbae@RP-002.local>
Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
authored and GitHub committed 2026-09-07 18:16:15 +09:00
1 parent 352ac2ffd1
commit ecd3192b40
5 files changed
+276 -8

No files matched your search

+132 -1
View File
@@ -1,5 +1,8 @@
import { describe, it, expect, vi } from 'vitest' import { describe, it, expect, vi } from 'vitest'
import { generateLiveSummary } from '@/lib/live-summary' import {
generateLiveSummary,
planLiveSummaryRequest,
} from '@/lib/live-summary'
describe('generateLiveSummary', () => { describe('generateLiveSummary', () => {
it('Gemini API 응답을 받아 중간 요약 마크다운을 반환한다', async () => { it('Gemini API 응답을 받아 중간 요약 마크다운을 반환한다', async () => {
@@ -155,3 +158,131 @@ describe('generateLiveSummary — openai-compatible', () => {
) )
}) })
}) })
describe('planLiveSummaryRequest', () => {
const base = {
totalChunks: 50,
lastSummarizedIndex: 30,
incrementsSinceFull: 3,
fullRefreshEvery: 20,
hasPreviousSummary: true,
}
it('평상시에는 직전 지점부터 증분으로 보낸다', () => {
expect(planLiveSummaryRequest(base)).toEqual({
mode: 'incremental',
startIndex: 30,
})
})
it('첫 호출은 전체를 보낸다', () => {
expect(
planLiveSummaryRequest({ ...base, lastSummarizedIndex: 0 }),
).toEqual({ mode: 'full', startIndex: 0 })
})
it('갱신할 직전 요약이 없으면 전체를 보낸다', () => {
expect(
planLiveSummaryRequest({ ...base, hasPreviousSummary: false }),
).toEqual({ mode: 'full', startIndex: 0 })
})
it('증분이 누적되면 전체 재요약으로 오차를 끊는다', () => {
expect(
planLiveSummaryRequest({ ...base, incrementsSinceFull: 20 }),
).toEqual({ mode: 'full', startIndex: 0 })
})
it('fullRefreshEvery=0이면 전체 재요약을 하지 않는다', () => {
expect(
planLiveSummaryRequest({
...base,
fullRefreshEvery: 0,
incrementsSinceFull: 999,
}).mode,
).toBe('incremental')
})
it('전사가 초기화되어 인덱스가 범위를 벗어나면 전체를 보낸다', () => {
expect(
planLiveSummaryRequest({ ...base, totalChunks: 5 }),
).toEqual({ mode: 'full', startIndex: 0 })
})
})
describe('generateLiveSummary — 증분 모드', () => {
function mockOk(text = '## 요약\n갱신됨') {
return vi.fn().mockResolvedValue({
ok: true,
json: () =>
Promise.resolve({
candidates: [{ content: { parts: [{ text }] } }],
}),
})
}
function sentPrompt(mockFetch: ReturnType<typeof vi.fn>): string {
const body = JSON.parse(mockFetch.mock.calls[0][1].body)
return body.contents[0].parts[0].text
}
it('previousSummary가 있으면 요약과 신규 발화를 나눠 전달한다', async () => {
const mockFetch = mockOk()
await generateLiveSummary('새로 나온 이야기', {
apiKey: 'k',
previousSummary: '## 요약\n이전까지의 내용',
fetchFn: mockFetch,
})
const prompt = sentPrompt(mockFetch)
expect(prompt).toContain('[지금까지의 요약]')
expect(prompt).toContain('이전까지의 내용')
expect(prompt).toContain('[새로 추가된 발화]')
expect(prompt).toContain('새로 나온 이야기')
expect(prompt).toContain('갱신')
})
it('previousSummary가 없으면 기존 전체 요약 형식을 유지한다', async () => {
const mockFetch = mockOk()
await generateLiveSummary('전체 전사', { apiKey: 'k', fetchFn: mockFetch })
const prompt = sentPrompt(mockFetch)
expect(prompt).toContain('음성 인식 텍스트:')
expect(prompt).not.toContain('[지금까지의 요약]')
})
it('빈 문자열 previousSummary는 증분으로 취급하지 않는다', async () => {
const mockFetch = mockOk()
await generateLiveSummary('전체 전사', {
apiKey: 'k',
previousSummary: ' ',
fetchFn: mockFetch,
})
expect(sentPrompt(mockFetch)).not.toContain('[지금까지의 요약]')
})
it('긴 회의에서 증분 프롬프트가 전체 프롬프트보다 짧다', async () => {
const longTranscript = '회의 발화 한 줄입니다.\n'.repeat(500)
const fullFetch = mockOk()
await generateLiveSummary(longTranscript, {
apiKey: 'k',
fetchFn: fullFetch,
})
const incFetch = mockOk()
await generateLiveSummary('마지막 30초에 나온 이야기', {
apiKey: 'k',
previousSummary: '## 요약\n지금까지의 요약 본문',
fetchFn: incFetch,
})
expect(sentPrompt(incFetch).length).toBeLessThan(
sentPrompt(fullFetch).length / 5,
)
})
})
+9 -1
View File
@@ -4,11 +4,12 @@ import { resolveProviderSettings } from '@/lib/api-keys'
import type { SummaryDepth, TemplateId } from '@/lib/templates' import type { SummaryDepth, TemplateId } from '@/lib/templates'
const MAX_TRANSCRIPT_CHARS = 40_000 const MAX_TRANSCRIPT_CHARS = 40_000
const MAX_PREVIOUS_SUMMARY_CHARS = 8_000
export async function POST(request: NextRequest) { export async function POST(request: NextRequest) {
try { try {
const body = await request.json() const body = await request.json()
const { transcript, template, depth, customPrompt } = body const { transcript, template, depth, customPrompt, previousSummary } = body
if (!transcript || typeof transcript !== 'string') { if (!transcript || typeof transcript !== 'string') {
return Response.json( return Response.json(
@@ -33,8 +34,15 @@ export async function POST(request: NextRequest) {
? transcript.slice(-MAX_TRANSCRIPT_CHARS) ? transcript.slice(-MAX_TRANSCRIPT_CHARS)
: transcript : transcript
// 증분 모드: 직전 요약 + 새 발화만 보내므로 호출당 토큰이 일정하다.
const previous =
typeof previousSummary === 'string'
? previousSummary.slice(-MAX_PREVIOUS_SUMMARY_CHARS)
: undefined
const result = await generateLiveSummary(truncated, { const result = await generateLiveSummary(truncated, {
provider: settings, provider: settings,
previousSummary: previous,
template: (template as TemplateId | undefined) ?? 'meeting', template: (template as TemplateId | undefined) ?? 'meeting',
depth: depth as SummaryDepth | undefined, depth: depth as SummaryDepth | undefined,
customPrompt: customPrompt:
+37 -3
View File
@@ -5,12 +5,15 @@ import type { TranscriptChunk } from '@/lib/transcript-formatter'
import { formatTranscriptChunks } from '@/lib/transcript-formatter' import { formatTranscriptChunks } from '@/lib/transcript-formatter'
import type { SummaryDepth, TemplateId } from '@/lib/templates' import type { SummaryDepth, TemplateId } from '@/lib/templates'
import { getProviderRequestPayload } from '@/lib/api-key-storage' import { getProviderRequestPayload } from '@/lib/api-key-storage'
import { planLiveSummaryRequest } from '@/lib/live-summary'
interface UseLiveSummaryOptions { interface UseLiveSummaryOptions {
enabled: boolean enabled: boolean
pollIntervalMs?: number pollIntervalMs?: number
minWords?: number minWords?: number
incrementWords?: number incrementWords?: number
/** 증분 요약을 이 횟수만큼 반복하면 전사 전체로 한 번 다시 요약한다. 0이면 끈다. */
fullRefreshEvery?: number
template?: TemplateId template?: TemplateId
depth?: SummaryDepth depth?: SummaryDepth
customPrompt?: string customPrompt?: string
@@ -41,6 +44,7 @@ export function useLiveSummary(
pollIntervalMs = 30_000, pollIntervalMs = 30_000,
minWords = 25, minWords = 25,
incrementWords = 40, incrementWords = 40,
fullRefreshEvery = 20,
template = 'meeting', template = 'meeting',
depth, depth,
customPrompt, customPrompt,
@@ -64,6 +68,11 @@ export function useLiveSummary(
const cooldownUntilRef = useRef<number | null>(null) const cooldownUntilRef = useRef<number | null>(null)
const consecutiveFailuresRef = useRef(0) const consecutiveFailuresRef = useRef(0)
// 증분 요약 상태 — 성공했을 때만 전진시켜서 실패해도 발화를 잃지 않는다.
const summaryRef = useRef('')
const lastSummarizedIndexRef = useRef(0)
const incrementsSinceFullRef = useRef(0)
useEffect(() => { useEffect(() => {
chunksRef.current = chunks chunksRef.current = chunks
}, [chunks]) }, [chunks])
@@ -98,12 +107,31 @@ export function useLiveSummary(
return return
} }
const transcript = formatTranscriptChunks(chunksRef.current) const allChunks = chunksRef.current
const transcript = formatTranscriptChunks(allChunks)
const currentWordCount = countWords(transcript) const currentWordCount = countWords(transcript)
if (currentWordCount < minWords) return if (currentWordCount < minWords) return
if (currentWordCount - lastWordCountRef.current < incrementWords) return if (currentWordCount - lastWordCountRef.current < incrementWords) return
const plan = planLiveSummaryRequest({
totalChunks: allChunks.length,
lastSummarizedIndex: lastSummarizedIndexRef.current,
incrementsSinceFull: incrementsSinceFullRef.current,
fullRefreshEvery,
hasPreviousSummary: summaryRef.current.trim().length > 0,
})
// 요청을 보내는 시점의 길이를 고정해 둔다. 응답을 기다리는 동안
// 새 청크가 쌓여도 그 부분은 다음 회차로 넘어간다.
const chunkCountAtSend = allChunks.length
const payloadTranscript =
plan.mode === 'full'
? transcript
: formatTranscriptChunks(allChunks.slice(plan.startIndex))
if (payloadTranscript.trim().length === 0) return
const controller = new AbortController() const controller = new AbortController()
abortRef.current = controller abortRef.current = controller
inFlightRef.current = true inFlightRef.current = true
@@ -115,7 +143,9 @@ export function useLiveSummary(
method: 'POST', method: 'POST',
headers: { 'Content-Type': 'application/json' }, headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ body: JSON.stringify({
transcript, transcript: payloadTranscript,
previousSummary:
plan.mode === 'incremental' ? summaryRef.current : undefined,
template: configRef.current.template, template: configRef.current.template,
depth: configRef.current.depth, depth: configRef.current.depth,
customPrompt: configRef.current.customPrompt, customPrompt: configRef.current.customPrompt,
@@ -142,8 +172,12 @@ export function useLiveSummary(
const data = await res.json() const data = await res.json()
setSummary(data.markdown) setSummary(data.markdown)
summaryRef.current = data.markdown
setLastUpdatedAt(Date.now()) setLastUpdatedAt(Date.now())
lastWordCountRef.current = currentWordCount lastWordCountRef.current = currentWordCount
lastSummarizedIndexRef.current = chunkCountAtSend
incrementsSinceFullRef.current =
plan.mode === 'full' ? 0 : incrementsSinceFullRef.current + 1
consecutiveFailuresRef.current = 0 consecutiveFailuresRef.current = 0
cooldownUntilRef.current = null cooldownUntilRef.current = null
setCooldownUntil(null) setCooldownUntil(null)
@@ -164,7 +198,7 @@ export function useLiveSummary(
return () => { return () => {
clearInterval(interval) clearInterval(interval)
} }
}, [enabled, pollIntervalMs, minWords, incrementWords]) }, [enabled, pollIntervalMs, minWords, incrementWords, fullRefreshEvery])
useEffect(() => { useEffect(() => {
return () => { return () => {
+66 -1
View File
@@ -14,6 +14,14 @@ interface LiveSummaryOptions {
/** 프로바이더 설정. 생략하면 apiKey로 Gemini를 호출한다. */ /** 프로바이더 설정. 생략하면 apiKey로 Gemini를 호출한다. */
provider?: ProviderSettings provider?: ProviderSettings
apiKey?: string apiKey?: string
/**
* 직전 롤링 요약.
*
* 값이 있으면 증분 모드로 동작하며, 이때 `transcript` 인자는 전사 전체가 아니라
* **직전 요약 이후 새로 추가된 발화**만 담아야 한다. 호출당 토큰이 회의 길이와
* 무관하게 일정해진다.
*/
previousSummary?: string
template?: TemplateId template?: TemplateId
depth?: SummaryDepth depth?: SummaryDepth
customPrompt?: string customPrompt?: string
@@ -32,13 +40,19 @@ export async function generateLiveSummary(
return { success: false, error: '요약할 텍스트가 비어 있습니다.' } return { success: false, error: '요약할 텍스트가 비어 있습니다.' }
} }
const previous = options.previousSummary?.trim() ?? ''
const incremental = previous.length > 0
const templateId = options.template ?? 'meeting' const templateId = options.template ?? 'meeting'
const depth = resolveDepth(templateId, options.depth) const depth = resolveDepth(templateId, options.depth)
const prompt = buildPrompt({ const prompt = buildPrompt({
templateId, templateId,
depth, depth,
transcript: trimmed, transcript: incremental
? `[지금까지의 요약]\n${previous}\n\n[새로 추가된 발화]\n${trimmed}`
: trimmed,
live: true, live: true,
incremental,
customPrompt: options.customPrompt, customPrompt: options.customPrompt,
}) })
@@ -58,3 +72,54 @@ export async function generateLiveSummary(
return { success: true, markdown: result.text } return { success: true, markdown: result.text }
} }
export interface LiveSummaryPlan {
/** 'full'이면 전사 전체를 보내고 previousSummary를 쓰지 않는다. */
mode: 'full' | 'incremental'
/** 이번에 보낼 청크의 시작 인덱스. full이면 0. */
startIndex: number
}
export interface LiveSummaryPlanArgs {
totalChunks: number
lastSummarizedIndex: number
incrementsSinceFull: number
/** 0이면 주기적 전체 재요약을 하지 않는다. */
fullRefreshEvery: number
hasPreviousSummary: boolean
}
/**
* 이번 롤링 요약 호출을 증분으로 보낼지 전체로 보낼지 결정한다.
*
* 증분 모드는 호출당 토큰을 회의 길이와 무관하게 유지하지만, 요약을 요약하는
* 구조라 반복될수록 오차가 쌓인다. 그래서 일정 횟수마다 전사 전체로 한 번씩
* 다시 요약해 오차를 끊는다.
*/
export function planLiveSummaryRequest(
args: LiveSummaryPlanArgs,
): LiveSummaryPlan {
const {
totalChunks,
lastSummarizedIndex,
incrementsSinceFull,
fullRefreshEvery,
hasPreviousSummary,
} = args
// 전사가 초기화되어 인덱스가 범위를 벗어난 경우
if (lastSummarizedIndex > totalChunks) {
return { mode: 'full', startIndex: 0 }
}
// 첫 호출이거나 갱신할 요약이 아직 없는 경우
if (lastSummarizedIndex === 0 || !hasPreviousSummary) {
return { mode: 'full', startIndex: 0 }
}
if (fullRefreshEvery > 0 && incrementsSinceFull >= fullRefreshEvery) {
return { mode: 'full', startIndex: 0 }
}
return { mode: 'incremental', startIndex: lastSummarizedIndex }
}
+32 -2
View File
@@ -197,11 +197,27 @@ const DEPTH_MODIFIERS: Record<SummaryDepth, string> = {
const LIVE_MODIFIER = `회의가 아직 진행 중입니다. 지금까지의 내용을 기반으로 **중간 정리**를 작성하세요. 확정되지 않은 결정은 "(논의 중)"으로 표시하세요.` const LIVE_MODIFIER = `회의가 아직 진행 중입니다. 지금까지의 내용을 기반으로 **중간 정리**를 작성하세요. 확정되지 않은 결정은 "(논의 중)"으로 표시하세요.`
/**
* 증분 갱신 모드.
*
* 전사 전체를 매번 다시 보내는 대신 [지금까지의 요약] + [새로 추가된 발화]만 보낸다.
* 호출당 토큰이 회의 길이와 무관하게 일정해져, 비용이 제곱이 아닌 선형으로 늘어난다.
*/
const INCREMENTAL_MODIFIER = `아래에는 [지금까지의 요약]과 [새로 추가된 발화]가 주어집니다.
기존 요약을 처음부터 다시 쓰지 말고 **갱신**하세요:
- 기존 요약의 내용과 구조를 유지한 채 새 발화를 반영합니다.
- 기존 항목이 새 발화로 확정되거나 번복되었다면 그 항목을 고치세요.
- 새 발화에 언급되지 않았다는 이유로 기존 내용을 삭제하지 마세요.
- 출력은 항상 갱신된 회의록 **전체**입니다. 변경분만 출력하지 마세요.`
export interface BuildPromptArgs { export interface BuildPromptArgs {
templateId: TemplateId templateId: TemplateId
depth: SummaryDepth depth: SummaryDepth
/** 증분 모드에서는 [지금까지의 요약] + [새로 추가된 발화]를 담은 블록이 들어온다. */
transcript: string transcript: string
live?: boolean live?: boolean
/** 직전 요약을 갱신하는 모드. live와 함께 쓴다. */
incremental?: boolean
customPrompt?: string customPrompt?: string
} }
@@ -222,7 +238,14 @@ export function resolveDepth(
} }
export function buildPrompt(args: BuildPromptArgs): string { export function buildPrompt(args: BuildPromptArgs): string {
const { templateId, depth, transcript, live = false, customPrompt } = args const {
templateId,
depth,
transcript,
live = false,
incremental = false,
customPrompt,
} = args
let instruction: string let instruction: string
@@ -243,12 +266,19 @@ export function buildPrompt(args: BuildPromptArgs): string {
if (live && template.liveSupported) { if (live && template.liveSupported) {
parts.push(LIVE_MODIFIER) parts.push(LIVE_MODIFIER)
if (incremental) {
parts.push(INCREMENTAL_MODIFIER)
}
} }
instruction = parts.join('\n\n') instruction = parts.join('\n\n')
} }
return `${instruction}\n\n---\n음성 인식 텍스트:\n${transcript}` // 증분 모드의 transcript는 자체 라벨([지금까지의 요약] 등)을 이미 포함한다.
const body = incremental ? transcript : `음성 인식 텍스트:\n${transcript}`
return `${instruction}\n\n---\n${body}`
} }
export function liveEnabledFor(templateId: TemplateId): boolean { export function liveEnabledFor(templateId: TemplateId): boolean {