perf: 롤링 요약을 증분 방식으로 전환해 토큰 비용을 선형화 (#15)

롤링 요약이 30초마다 누적 전사 전체를 다시 보내고 있었다. 호출 N회차가
그때까지의 전사 전부를 담으므로 총 토큰이 회의 길이의 제곱으로 늘어난다.
1시간 회의 기준 약 116만 입력 토큰, 2시간이면 2배가 아니라 4배가 된다.

[지금까지의 요약] + [새로 추가된 발화]만 보내도록 바꿔 호출당 토큰을
회의 길이와 무관하게 일정하게 만들었다.

- templates.ts: BuildPromptArgs.incremental 추가, INCREMENTAL_MODIFIER 신설.
  증분 모드에서는 전사 블록이 자체 라벨을 갖는다
- live-summary.ts: previousSummary 옵션. 값이 있으면 증분 모드로 동작
- live-summary.ts: planLiveSummaryRequest() — 전체/증분 결정을 순수 함수로
  분리해 단위 테스트 가능하게 함
- useLiveSummary: 마지막 요약 지점 인덱스를 추적해 델타만 전송.
  성공했을 때만 전진시켜 실패해도 발화를 잃지 않는다
- 요약을 요약하는 구조라 오차가 누적되므로 fullRefreshEvery(기본 20회)마다
  전사 전체로 한 번 다시 요약해 오차를 끊는다

기본 설정에서 1시간 회의 기준 약 7배, 전체 재요약을 끄면 약 12배 절감된다.
회의가 2~3분보다 짧으면 지시문 오버헤드 때문에 오히려 조금 늘어난다.

테스트 141 → 151.

Co-authored-by: csbae <csbae@RP-002.local>
Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
authored and GitHub committed 2026-09-07 18:16:15 +09:00
1 parent 352ac2ffd1
commit ecd3192b40
5 files changed
+276 -8

No files matched your search

+66 -1
View File
@@ -14,6 +14,14 @@ interface LiveSummaryOptions {
/** 프로바이더 설정. 생략하면 apiKey로 Gemini를 호출한다. */
provider?: ProviderSettings
apiKey?: string
/**
* 직전 롤링 요약.
*
* 값이 있으면 증분 모드로 동작하며, 이때 `transcript` 인자는 전사 전체가 아니라
* **직전 요약 이후 새로 추가된 발화**만 담아야 한다. 호출당 토큰이 회의 길이와
* 무관하게 일정해진다.
*/
previousSummary?: string
template?: TemplateId
depth?: SummaryDepth
customPrompt?: string
@@ -32,13 +40,19 @@ export async function generateLiveSummary(
return { success: false, error: '요약할 텍스트가 비어 있습니다.' }
}
const previous = options.previousSummary?.trim() ?? ''
const incremental = previous.length > 0
const templateId = options.template ?? 'meeting'
const depth = resolveDepth(templateId, options.depth)
const prompt = buildPrompt({
templateId,
depth,
transcript: trimmed,
transcript: incremental
? `[지금까지의 요약]\n${previous}\n\n[새로 추가된 발화]\n${trimmed}`
: trimmed,
live: true,
incremental,
customPrompt: options.customPrompt,
})
@@ -58,3 +72,54 @@ export async function generateLiveSummary(
return { success: true, markdown: result.text }
}
export interface LiveSummaryPlan {
/** 'full'이면 전사 전체를 보내고 previousSummary를 쓰지 않는다. */
mode: 'full' | 'incremental'
/** 이번에 보낼 청크의 시작 인덱스. full이면 0. */
startIndex: number
}
export interface LiveSummaryPlanArgs {
totalChunks: number
lastSummarizedIndex: number
incrementsSinceFull: number
/** 0이면 주기적 전체 재요약을 하지 않는다. */
fullRefreshEvery: number
hasPreviousSummary: boolean
}
/**
* 이번 롤링 요약 호출을 증분으로 보낼지 전체로 보낼지 결정한다.
*
* 증분 모드는 호출당 토큰을 회의 길이와 무관하게 유지하지만, 요약을 요약하는
* 구조라 반복될수록 오차가 쌓인다. 그래서 일정 횟수마다 전사 전체로 한 번씩
* 다시 요약해 오차를 끊는다.
*/
export function planLiveSummaryRequest(
args: LiveSummaryPlanArgs,
): LiveSummaryPlan {
const {
totalChunks,
lastSummarizedIndex,
incrementsSinceFull,
fullRefreshEvery,
hasPreviousSummary,
} = args
// 전사가 초기화되어 인덱스가 범위를 벗어난 경우
if (lastSummarizedIndex > totalChunks) {
return { mode: 'full', startIndex: 0 }
}
// 첫 호출이거나 갱신할 요약이 아직 없는 경우
if (lastSummarizedIndex === 0 || !hasPreviousSummary) {
return { mode: 'full', startIndex: 0 }
}
if (fullRefreshEvery > 0 && incrementsSinceFull >= fullRefreshEvery) {
return { mode: 'full', startIndex: 0 }
}
return { mode: 'incremental', startIndex: lastSummarizedIndex }
}
+32 -2
View File
@@ -197,11 +197,27 @@ const DEPTH_MODIFIERS: Record<SummaryDepth, string> = {
const LIVE_MODIFIER = `회의가 아직 진행 중입니다. 지금까지의 내용을 기반으로 **중간 정리**를 작성하세요. 확정되지 않은 결정은 "(논의 중)"으로 표시하세요.`
/**
* 증분 갱신 모드.
*
* 전사 전체를 매번 다시 보내는 대신 [지금까지의 요약] + [새로 추가된 발화]만 보낸다.
* 호출당 토큰이 회의 길이와 무관하게 일정해져, 비용이 제곱이 아닌 선형으로 늘어난다.
*/
const INCREMENTAL_MODIFIER = `아래에는 [지금까지의 요약]과 [새로 추가된 발화]가 주어집니다.
기존 요약을 처음부터 다시 쓰지 말고 **갱신**하세요:
- 기존 요약의 내용과 구조를 유지한 채 새 발화를 반영합니다.
- 기존 항목이 새 발화로 확정되거나 번복되었다면 그 항목을 고치세요.
- 새 발화에 언급되지 않았다는 이유로 기존 내용을 삭제하지 마세요.
- 출력은 항상 갱신된 회의록 **전체**입니다. 변경분만 출력하지 마세요.`
export interface BuildPromptArgs {
templateId: TemplateId
depth: SummaryDepth
/** 증분 모드에서는 [지금까지의 요약] + [새로 추가된 발화]를 담은 블록이 들어온다. */
transcript: string
live?: boolean
/** 직전 요약을 갱신하는 모드. live와 함께 쓴다. */
incremental?: boolean
customPrompt?: string
}
@@ -222,7 +238,14 @@ export function resolveDepth(
}
export function buildPrompt(args: BuildPromptArgs): string {
const { templateId, depth, transcript, live = false, customPrompt } = args
const {
templateId,
depth,
transcript,
live = false,
incremental = false,
customPrompt,
} = args
let instruction: string
@@ -243,12 +266,19 @@ export function buildPrompt(args: BuildPromptArgs): string {
if (live && template.liveSupported) {
parts.push(LIVE_MODIFIER)
if (incremental) {
parts.push(INCREMENTAL_MODIFIER)
}
}
instruction = parts.join('\n\n')
}
return `${instruction}\n\n---\n음성 인식 텍스트:\n${transcript}`
// 증분 모드의 transcript는 자체 라벨([지금까지의 요약] 등)을 이미 포함한다.
const body = incremental ? transcript : `음성 인식 텍스트:\n${transcript}`
return `${instruction}\n\n---\n${body}`
}
export function liveEnabledFor(templateId: TemplateId): boolean {