|
|
|
@@ -4,6 +4,7 @@ import { Session } from "."
|
|
|
|
|
import { SessionID, MessageID, PartID } from "./schema"
|
|
|
|
|
import { Instance } from "../project/instance"
|
|
|
|
|
import { Provider } from "../provider/provider"
|
|
|
|
|
import { ProviderTransform } from "../provider/transform"
|
|
|
|
|
import { MessageV2 } from "./message-v2"
|
|
|
|
|
import z from "zod"
|
|
|
|
|
import { Token } from "../util/token"
|
|
|
|
@@ -35,6 +36,24 @@ export namespace SessionCompaction {
|
|
|
|
|
export const PRUNE_MINIMUM = 20_000
|
|
|
|
|
export const PRUNE_PROTECT = 40_000
|
|
|
|
|
const PRUNE_PROTECTED_TOOLS = ["skill"]
|
|
|
|
|
const DEFAULT_TAIL_TURNS = 2
|
|
|
|
|
const MIN_TAIL_TOKENS = 2_000
|
|
|
|
|
const MAX_TAIL_TOKENS = 8_000
|
|
|
|
|
|
|
|
|
|
function usable(input: { cfg: Config.Info; model: Provider.Model }) {
|
|
|
|
|
const reserved =
|
|
|
|
|
input.cfg.compaction?.reserved ?? Math.min(20_000, ProviderTransform.maxOutputTokens(input.model))
|
|
|
|
|
return input.model.limit.input
|
|
|
|
|
? Math.max(0, input.model.limit.input - reserved)
|
|
|
|
|
: Math.max(0, input.model.limit.context - ProviderTransform.maxOutputTokens(input.model))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function tailBudget(input: { cfg: Config.Info; model: Provider.Model }) {
|
|
|
|
|
return (
|
|
|
|
|
input.cfg.compaction?.tail_tokens ??
|
|
|
|
|
Math.min(MAX_TAIL_TOKENS, Math.max(MIN_TAIL_TOKENS, Math.floor(usable(input) * 0.25)))
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
export interface Interface {
|
|
|
|
|
readonly isOverflow: (input: {
|
|
|
|
@@ -88,6 +107,55 @@ export namespace SessionCompaction {
|
|
|
|
|
return overflow({ cfg: yield* config.get(), tokens: input.tokens, model: input.model })
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
const estimate = Effect.fn("SessionCompaction.estimate")(function* (input: {
|
|
|
|
|
messages: MessageV2.WithParts[]
|
|
|
|
|
model: Provider.Model
|
|
|
|
|
}) {
|
|
|
|
|
const msgs = yield* MessageV2.toModelMessagesEffect(input.messages, input.model, { stripMedia: true })
|
|
|
|
|
return Token.estimate(JSON.stringify(msgs))
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
const select = Effect.fn("SessionCompaction.select")(function* (input: {
|
|
|
|
|
messages: MessageV2.WithParts[]
|
|
|
|
|
cfg: Config.Info
|
|
|
|
|
model: Provider.Model
|
|
|
|
|
}) {
|
|
|
|
|
const limit = input.cfg.compaction?.tail_turns ?? DEFAULT_TAIL_TURNS
|
|
|
|
|
if (limit <= 0) return { head: input.messages, tail_start_id: undefined }
|
|
|
|
|
const budget = tailBudget({ cfg: input.cfg, model: input.model })
|
|
|
|
|
const turns = input.messages.flatMap((msg, idx) =>
|
|
|
|
|
msg.info.role === "user" && !msg.parts.some((part) => part.type === "compaction") ? [idx] : [],
|
|
|
|
|
)
|
|
|
|
|
if (!turns.length) return { head: input.messages, tail_start_id: undefined }
|
|
|
|
|
|
|
|
|
|
let total = 0
|
|
|
|
|
let start = input.messages.length
|
|
|
|
|
let kept = 0
|
|
|
|
|
|
|
|
|
|
for (let i = turns.length - 1; i >= 0 && kept < limit; i--) {
|
|
|
|
|
const idx = turns[i]
|
|
|
|
|
const end = i + 1 < turns.length ? turns[i + 1] : input.messages.length
|
|
|
|
|
const size = yield* estimate({
|
|
|
|
|
messages: input.messages.slice(idx, end),
|
|
|
|
|
model: input.model,
|
|
|
|
|
})
|
|
|
|
|
if (kept === 0 && size > budget) {
|
|
|
|
|
log.info("tail fallback", { budget, size })
|
|
|
|
|
return { head: input.messages, tail_start_id: undefined }
|
|
|
|
|
}
|
|
|
|
|
if (total + size > budget) break
|
|
|
|
|
total += size
|
|
|
|
|
start = idx
|
|
|
|
|
kept++
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (kept === 0 || start === 0) return { head: input.messages, tail_start_id: undefined }
|
|
|
|
|
return {
|
|
|
|
|
head: input.messages.slice(0, start),
|
|
|
|
|
tail_start_id: input.messages[start]?.info.id,
|
|
|
|
|
}
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
// goes backwards through parts until there are PRUNE_PROTECT tokens worth of tool
|
|
|
|
|
// calls, then erases output of older tool calls to free context space
|
|
|
|
|
const prune = Effect.fn("SessionCompaction.prune")(function* (input: { sessionID: SessionID }) {
|
|
|
|
@@ -150,6 +218,7 @@ export namespace SessionCompaction {
|
|
|
|
|
throw new Error(`Compaction parent must be a user message: ${input.parentID}`)
|
|
|
|
|
}
|
|
|
|
|
const userMessage = parent.info
|
|
|
|
|
const compactionPart = parent.parts.find((part): part is MessageV2.CompactionPart => part.type === "compaction")
|
|
|
|
|
|
|
|
|
|
let messages = input.messages
|
|
|
|
|
let replay:
|
|
|
|
@@ -180,14 +249,22 @@ export namespace SessionCompaction {
|
|
|
|
|
const model = agent.model
|
|
|
|
|
? yield* provider.getModel(agent.model.providerID, agent.model.modelID)
|
|
|
|
|
: yield* provider.getModel(userMessage.model.providerID, userMessage.model.modelID)
|
|
|
|
|
const cfg = yield* config.get()
|
|
|
|
|
const history = compactionPart && messages.at(-1)?.info.id === input.parentID ? messages.slice(0, -1) : messages
|
|
|
|
|
const selected = yield* select({
|
|
|
|
|
messages: history,
|
|
|
|
|
cfg,
|
|
|
|
|
model,
|
|
|
|
|
})
|
|
|
|
|
// Allow plugins to inject context or replace compaction prompt.
|
|
|
|
|
const compacting = yield* plugin.trigger(
|
|
|
|
|
"experimental.session.compacting",
|
|
|
|
|
{ sessionID: input.sessionID },
|
|
|
|
|
{ context: [], prompt: undefined },
|
|
|
|
|
)
|
|
|
|
|
const defaultPrompt = `Provide a detailed prompt for continuing our conversation above.
|
|
|
|
|
Focus on information that would be helpful for continuing the conversation, including what we did, what we're doing, which files we're working on, and what we're going to do next.
|
|
|
|
|
const defaultPrompt = `Summarize the older conversation history so another agent can continue the work with the retained recent turns.
|
|
|
|
|
The most recent conversation turns will remain verbatim outside this summary, so focus on older context that is still needed to understand and continue the work.
|
|
|
|
|
Include what we did, what we're doing, which files we're working on, and what we're going to do next.
|
|
|
|
|
The summary that you construct will be used so that another agent can read it and continue the work.
|
|
|
|
|
Do not call any tools. Respond only with the summary text.
|
|
|
|
|
Respond in the same language as the user's messages in the conversation.
|
|
|
|
@@ -217,7 +294,7 @@ When constructing the summary, try to stick to this template:
|
|
|
|
|
---`
|
|
|
|
|
|
|
|
|
|
const prompt = compacting.prompt ?? [defaultPrompt, ...compacting.context].join("\n\n")
|
|
|
|
|
const msgs = structuredClone(messages)
|
|
|
|
|
const msgs = structuredClone(selected.head)
|
|
|
|
|
yield* plugin.trigger("experimental.chat.messages.transform", {}, { messages: msgs })
|
|
|
|
|
const modelMessages = yield* MessageV2.toModelMessagesEffect(msgs, model, { stripMedia: true })
|
|
|
|
|
const ctx = yield* InstanceState.context
|
|
|
|
@@ -280,6 +357,13 @@ When constructing the summary, try to stick to this template:
|
|
|
|
|
return "stop"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (compactionPart && selected.tail_start_id && compactionPart.tail_start_id !== selected.tail_start_id) {
|
|
|
|
|
yield* session.updatePart({
|
|
|
|
|
...compactionPart,
|
|
|
|
|
tail_start_id: selected.tail_start_id,
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (result === "continue" && input.auto) {
|
|
|
|
|
if (replay) {
|
|
|
|
|
const original = replay.info
|
|
|
|
|