compaction.test.ts 62 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819
  1. import { afterEach, describe, expect, mock, test } from "bun:test"
  2. import { ConfigV1 } from "@kirincode-ai/core/v1/config/config"
  3. import { SessionV1 } from "@kirincode-ai/core/v1/session"
  4. import { Database } from "@kirincode-ai/core/database/database"
  5. import { EventV2Bridge } from "@/event-v2-bridge"
  6. import { APICallError } from "ai"
  7. import { Cause, Deferred, Effect, Exit, Fiber, Layer, Schema } from "effect"
  8. import * as Stream from "effect/Stream"
  9. import { Config } from "@/config/config"
  10. import { LLM } from "../../src/session/llm"
  11. import { SessionCompaction } from "../../src/session/compaction"
  12. import { Token } from "@/util/token"
  13. import { Plugin } from "../../src/plugin"
  14. import { provideTmpdirInstance, TestInstance } from "../fixture/fixture"
  15. import { Session as SessionNs } from "@/session/session"
  16. import { MessageV2 } from "../../src/session/message-v2"
  17. import { MessageID, PartID, SessionID } from "../../src/session/schema"
  18. import { SessionStatus } from "../../src/session/status"
  19. import { SessionSummary } from "../../src/session/summary"
  20. import { SessionV2 } from "@kirincode-ai/core/session"
  21. import { SessionExecution } from "@kirincode-ai/core/session/execution"
  22. import { SessionProjector } from "@kirincode-ai/core/session/projector"
  23. import { Provider } from "@/provider/provider"
  24. import * as SessionProcessorModule from "../../src/session/processor"
  25. import { ProviderTest } from "../fake/provider"
  26. import { testEffect } from "../lib/effect"
  27. import { CrossSpawnSpawner } from "@kirincode-ai/core/cross-spawn-spawner"
  28. import { TestConfig } from "../fixture/config"
  29. import { RuntimeFlags } from "@/effect/runtime-flags"
  30. import { LLMEvent, Usage } from "@kirincode-ai/llm"
  31. import { ProviderV2 } from "@kirincode-ai/core/provider"
  32. import { ModelV2 } from "@kirincode-ai/core/model"
  33. import { AppNodeBuilder } from "@kirincode-ai/core/effect/app-node-builder"
  34. import { LayerNode } from "@kirincode-ai/core/effect/layer-node"
  35. const summary = Layer.succeed(
  36. SessionSummary.Service,
  37. SessionSummary.Service.of({
  38. summarize: () => Effect.void,
  39. diff: () => Effect.succeed([]),
  40. computeDiff: () => Effect.succeed([]),
  41. }),
  42. )
  43. const ref = {
  44. providerID: ProviderV2.ID.make("test"),
  45. modelID: ModelV2.ID.make("test-model"),
  46. }
  47. const usage = (input: ConstructorParameters<typeof Usage>[0]) => new Usage(input)
  48. const basicUsage = () => usage({ inputTokens: 1, outputTokens: 1, totalTokens: 2 })
  49. afterEach(() => {
  50. mock.restore()
  51. })
  52. function createModel(opts: {
  53. context: number
  54. output: number
  55. input?: number
  56. cost?: Provider.Model["cost"]
  57. npm?: string
  58. }): Provider.Model {
  59. return {
  60. id: "test-model",
  61. providerID: "test",
  62. name: "Test",
  63. limit: {
  64. context: opts.context,
  65. input: opts.input,
  66. output: opts.output,
  67. },
  68. cost: opts.cost ?? { input: 0, output: 0, cache: { read: 0, write: 0 } },
  69. capabilities: {
  70. toolcall: true,
  71. attachment: false,
  72. reasoning: false,
  73. temperature: true,
  74. input: { text: true, image: false, audio: false, video: false },
  75. output: { text: true, image: false, audio: false, video: false },
  76. },
  77. api: { npm: opts.npm ?? "@ai-sdk/anthropic" },
  78. options: {},
  79. } as Provider.Model
  80. }
  81. const wide = () => ProviderTest.fake({ model: createModel({ context: 100_000, output: 32_000 }) })
  82. function createUserMessage(sessionID: SessionID, text: string) {
  83. return Effect.gen(function* () {
  84. const ssn = yield* SessionNs.Service
  85. const msg = yield* ssn.updateMessage({
  86. id: MessageID.ascending(),
  87. role: "user",
  88. sessionID,
  89. agent: "build",
  90. model: ref,
  91. time: { created: Date.now() },
  92. })
  93. yield* ssn.updatePart({
  94. id: PartID.ascending(),
  95. messageID: msg.id,
  96. sessionID,
  97. type: "text",
  98. text,
  99. })
  100. return msg
  101. })
  102. }
  103. function createAssistantMessage(sessionID: SessionID, parentID: MessageID, root: string) {
  104. return SessionNs.Service.use((ssn) =>
  105. ssn.updateMessage({
  106. id: MessageID.ascending(),
  107. role: "assistant",
  108. sessionID,
  109. mode: "build",
  110. agent: "build",
  111. path: { cwd: root, root },
  112. cost: 0,
  113. tokens: {
  114. output: 0,
  115. input: 0,
  116. reasoning: 0,
  117. cache: { read: 0, write: 0 },
  118. },
  119. modelID: ref.modelID,
  120. providerID: ref.providerID,
  121. parentID,
  122. time: { created: Date.now() },
  123. finish: "end_turn",
  124. }),
  125. )
  126. }
  127. function createSummaryAssistantMessage(sessionID: SessionID, parentID: MessageID, root: string, text: string) {
  128. return SessionNs.Service.use((ssn) =>
  129. Effect.gen(function* () {
  130. const msg = yield* ssn.updateMessage({
  131. id: MessageID.ascending(),
  132. role: "assistant",
  133. sessionID,
  134. mode: "compaction",
  135. agent: "compaction",
  136. path: { cwd: root, root },
  137. cost: 0,
  138. tokens: {
  139. output: 0,
  140. input: 0,
  141. reasoning: 0,
  142. cache: { read: 0, write: 0 },
  143. },
  144. modelID: ref.modelID,
  145. providerID: ref.providerID,
  146. parentID,
  147. summary: true,
  148. time: { created: Date.now() },
  149. finish: "end_turn",
  150. })
  151. yield* ssn.updatePart({
  152. id: PartID.ascending(),
  153. messageID: msg.id,
  154. sessionID,
  155. type: "text",
  156. text,
  157. })
  158. return msg
  159. }),
  160. )
  161. }
  162. function createCompactionMarker(sessionID: SessionID) {
  163. return SessionNs.Service.use((ssn) =>
  164. Effect.gen(function* () {
  165. const msg = yield* ssn.updateMessage({
  166. id: MessageID.ascending(),
  167. role: "user",
  168. model: ref,
  169. sessionID,
  170. agent: "build",
  171. time: { created: Date.now() },
  172. })
  173. yield* ssn.updatePart({
  174. id: PartID.ascending(),
  175. messageID: msg.id,
  176. sessionID: msg.sessionID,
  177. type: "compaction",
  178. auto: false,
  179. })
  180. }),
  181. )
  182. }
  183. function fake(
  184. input: Parameters<SessionProcessorModule.SessionProcessor.Interface["create"]>[0],
  185. result: "continue" | "compact",
  186. ) {
  187. const msg = input.assistantMessage
  188. return {
  189. get message() {
  190. return msg
  191. },
  192. updateToolCall: Effect.fn("TestSessionProcessor.updateToolCall")(() => Effect.succeed(undefined)),
  193. completeToolCall: Effect.fn("TestSessionProcessor.completeToolCall")(() => Effect.void),
  194. process: Effect.fn("TestSessionProcessor.process")(() => Effect.succeed(result)),
  195. } satisfies SessionProcessorModule.SessionProcessor.Handle
  196. }
  197. function processorLayer(result: "continue" | "compact") {
  198. return Layer.succeed(
  199. SessionProcessorModule.SessionProcessor.Service,
  200. SessionProcessorModule.SessionProcessor.Service.of({
  201. create: Effect.fn("TestSessionProcessor.create")((input) => Effect.succeed(fake(input, result))),
  202. }),
  203. )
  204. }
  205. function cfg(compaction?: ConfigV1.Info["compaction"]) {
  206. const base = Schema.decodeUnknownSync(ConfigV1.Info)({}) as ConfigV1.Info
  207. return Layer.succeed(Config.Service, TestConfig.make({ get: () => Effect.succeed({ ...base, compaction }) }))
  208. }
  209. const defaultProvider = wide()
  210. const compactionTestNode = LayerNode.group([
  211. SessionCompaction.node,
  212. SessionNs.node,
  213. SessionProjector.node,
  214. Database.node,
  215. EventV2Bridge.node,
  216. CrossSpawnSpawner.node,
  217. ])
  218. const env = AppNodeBuilder.build(compactionTestNode, [
  219. [Provider.node, defaultProvider.layer],
  220. [SessionProcessorModule.SessionProcessor.node, processorLayer("continue")],
  221. [RuntimeFlags.node, RuntimeFlags.layer({ experimentalEventSystem: true })],
  222. ])
  223. const it = testEffect(env)
  224. const compactionEnv = AppNodeBuilder.build(
  225. LayerNode.group([SessionNs.node, SessionProjector.node, Database.node, EventV2Bridge.node, CrossSpawnSpawner.node]),
  226. )
  227. const itCompaction = testEffect(compactionEnv)
  228. type CompactionProcessOptions = {
  229. result?: "continue" | "compact"
  230. llm?: Layer.Layer<LLM.Service>
  231. plugin?: Layer.Layer<Plugin.Service>
  232. provider?: ReturnType<typeof wide>
  233. config?: Layer.Layer<Config.Service>
  234. }
  235. function withCompaction(options?: CompactionProcessOptions) {
  236. return Effect.provide(compactionProcessLayer(options))
  237. }
  238. function compactionProcessLayer(options?: CompactionProcessOptions) {
  239. const replacements: LayerNode.Replacements = [
  240. [Provider.node, (options?.provider ?? wide()).layer],
  241. [RuntimeFlags.node, RuntimeFlags.layer({ experimentalEventSystem: true })],
  242. [SessionSummary.node, summary],
  243. ]
  244. if (!options?.llm) {
  245. return AppNodeBuilder.build(compactionTestNode, [
  246. ...replacements,
  247. [SessionProcessorModule.SessionProcessor.node, processorLayer(options?.result ?? "continue")],
  248. ...(options?.plugin ? ([[Plugin.node, options.plugin]] as const) : []),
  249. ...(options?.config ? ([[Config.node, options.config]] as const) : []),
  250. ])
  251. }
  252. return AppNodeBuilder.build(compactionTestNode, [
  253. ...replacements,
  254. [LLM.node, options.llm],
  255. ...(options?.plugin ? ([[Plugin.node, options.plugin]] as const) : []),
  256. ...(options?.config ? ([[Config.node, options.config]] as const) : []),
  257. ])
  258. }
  259. function createSummaryCompaction(sessionID: SessionID) {
  260. return SessionCompaction.use.create({ sessionID, agent: "build", model: ref, auto: false })
  261. }
  262. function readCompactionPart(sessionID: SessionID) {
  263. return SessionNs.use
  264. .messages({ sessionID })
  265. .pipe(
  266. Effect.map((messages) =>
  267. messages.at(-2)?.parts.find((item): item is SessionV1.CompactionPart => item.type === "compaction"),
  268. ),
  269. )
  270. }
  271. function llm() {
  272. const queue: Array<
  273. Stream.Stream<LLMEvent, unknown> | ((input: LLM.StreamInput) => Stream.Stream<LLMEvent, unknown>)
  274. > = []
  275. return {
  276. push(stream: Stream.Stream<LLMEvent, unknown> | ((input: LLM.StreamInput) => Stream.Stream<LLMEvent, unknown>)) {
  277. queue.push(stream)
  278. },
  279. llmLayer: Layer.succeed(
  280. LLM.Service,
  281. LLM.Service.of({
  282. stream: (input) => {
  283. const item = queue.shift() ?? Stream.empty
  284. const stream = typeof item === "function" ? item(input) : item
  285. return stream.pipe(Stream.mapEffect((event) => Effect.succeed(event)))
  286. },
  287. }),
  288. ),
  289. }
  290. }
  291. function reply(
  292. text: string,
  293. capture?: (input: LLM.StreamInput) => void,
  294. ): (input: LLM.StreamInput) => Stream.Stream<LLMEvent, unknown> {
  295. return (input) => {
  296. capture?.(input)
  297. return Stream.make(
  298. LLMEvent.textStart({ id: "txt-0" }),
  299. LLMEvent.textDelta({ id: "txt-0", text }),
  300. LLMEvent.textEnd({ id: "txt-0" }),
  301. LLMEvent.stepFinish({
  302. index: 0,
  303. reason: "stop",
  304. usage: basicUsage(),
  305. }),
  306. LLMEvent.finish({
  307. reason: "stop",
  308. usage: basicUsage(),
  309. }),
  310. )
  311. }
  312. }
  313. function plugin(ready: Deferred.Deferred<void>) {
  314. return Layer.mock(Plugin.Service)({
  315. trigger: <Name extends string, Input, Output>(name: Name, _input: Input, output: Output) => {
  316. if (name !== "experimental.session.compacting") return Effect.succeed(output)
  317. return Effect.sync(() => Deferred.doneUnsafe(ready, Effect.void)).pipe(
  318. Effect.andThen(Effect.never),
  319. Effect.as(output),
  320. )
  321. },
  322. list: () => Effect.succeed([]),
  323. init: () => Effect.void,
  324. })
  325. }
  326. function autocontinue(enabled: boolean) {
  327. return Layer.mock(Plugin.Service)({
  328. trigger: <Name extends string, Input, Output>(name: Name, _input: Input, output: Output) => {
  329. if (name !== "experimental.compaction.autocontinue") return Effect.succeed(output)
  330. return Effect.sync(() => {
  331. ;(output as { enabled: boolean }).enabled = enabled
  332. return output
  333. })
  334. },
  335. list: () => Effect.succeed([]),
  336. init: () => Effect.void,
  337. })
  338. }
  339. describe("session.compaction.isOverflow", () => {
  340. it.live(
  341. "returns true when token count exceeds usable context",
  342. provideTmpdirInstance(() =>
  343. Effect.gen(function* () {
  344. const compact = yield* SessionCompaction.Service
  345. const model = createModel({ context: 100_000, output: 32_000 })
  346. const tokens = { input: 75_000, output: 5_000, reasoning: 0, cache: { read: 0, write: 0 } }
  347. expect(yield* compact.isOverflow({ tokens, model })).toBe(true)
  348. }),
  349. ),
  350. )
  351. it.live(
  352. "returns false when token count within usable context",
  353. provideTmpdirInstance(() =>
  354. Effect.gen(function* () {
  355. const compact = yield* SessionCompaction.Service
  356. const model = createModel({ context: 200_000, output: 32_000 })
  357. const tokens = { input: 100_000, output: 10_000, reasoning: 0, cache: { read: 0, write: 0 } }
  358. expect(yield* compact.isOverflow({ tokens, model })).toBe(false)
  359. }),
  360. ),
  361. )
  362. it.live(
  363. "includes cache.read in token count",
  364. provideTmpdirInstance(() =>
  365. Effect.gen(function* () {
  366. const compact = yield* SessionCompaction.Service
  367. const model = createModel({ context: 100_000, output: 32_000 })
  368. const tokens = { input: 60_000, output: 10_000, reasoning: 0, cache: { read: 10_000, write: 0 } }
  369. expect(yield* compact.isOverflow({ tokens, model })).toBe(true)
  370. }),
  371. ),
  372. )
  373. it.live(
  374. "respects input limit for input caps",
  375. provideTmpdirInstance(() =>
  376. Effect.gen(function* () {
  377. const compact = yield* SessionCompaction.Service
  378. const model = createModel({ context: 400_000, input: 272_000, output: 128_000 })
  379. const tokens = { input: 271_000, output: 1_000, reasoning: 0, cache: { read: 2_000, write: 0 } }
  380. expect(yield* compact.isOverflow({ tokens, model })).toBe(true)
  381. }),
  382. ),
  383. )
  384. it.live(
  385. "returns false when input/output are within input caps",
  386. provideTmpdirInstance(() =>
  387. Effect.gen(function* () {
  388. const compact = yield* SessionCompaction.Service
  389. const model = createModel({ context: 400_000, input: 272_000, output: 128_000 })
  390. const tokens = { input: 200_000, output: 20_000, reasoning: 0, cache: { read: 10_000, write: 0 } }
  391. expect(yield* compact.isOverflow({ tokens, model })).toBe(false)
  392. }),
  393. ),
  394. )
  395. it.live(
  396. "returns false when output within limit with input caps",
  397. provideTmpdirInstance(() =>
  398. Effect.gen(function* () {
  399. const compact = yield* SessionCompaction.Service
  400. const model = createModel({ context: 200_000, input: 120_000, output: 10_000 })
  401. const tokens = { input: 50_000, output: 9_999, reasoning: 0, cache: { read: 0, write: 0 } }
  402. expect(yield* compact.isOverflow({ tokens, model })).toBe(false)
  403. }),
  404. ),
  405. )
  406. // ─── Bug reproduction tests ───────────────────────────────────────────
  407. // These tests demonstrate that when limit.input is set, isOverflow()
  408. // does not subtract any headroom for the next model response. This means
  409. // compaction only triggers AFTER we've already consumed the full input
  410. // budget, leaving zero room for the next API call's output tokens.
  411. //
  412. // Compare: without limit.input, usable = context - output (reserves space).
  413. // With limit.input, usable = limit.input (reserves nothing).
  414. //
  415. // Related issues: #10634, #8089, #11086, #12621
  416. // Open PRs: #6875, #12924
  417. it.live(
  418. "BUG: no headroom when limit.input is set — compaction should trigger near boundary but does not",
  419. provideTmpdirInstance(() =>
  420. Effect.gen(function* () {
  421. const compact = yield* SessionCompaction.Service
  422. // Simulate Claude with prompt caching: input limit = 200K, output limit = 32K
  423. const model = createModel({ context: 200_000, input: 200_000, output: 32_000 })
  424. // We've used 198K tokens total. Only 2K under the input limit.
  425. // On the next turn, the full conversation (198K) becomes input,
  426. // plus the model needs room to generate output — this WILL overflow.
  427. const tokens = { input: 180_000, output: 15_000, reasoning: 0, cache: { read: 3_000, write: 0 } }
  428. // count = 180K + 3K + 15K = 198K
  429. // usable = limit.input = 200K (no output subtracted!)
  430. // 198K > 200K = false → no compaction triggered
  431. // WITHOUT limit.input: usable = 200K - 32K = 168K, and 198K > 168K = true ✓
  432. // WITH limit.input: usable = 200K, and 198K > 200K = false ✗
  433. // With 198K used and only 2K headroom, the next turn will overflow.
  434. // Compaction MUST trigger here.
  435. expect(yield* compact.isOverflow({ tokens, model })).toBe(true)
  436. }),
  437. ),
  438. )
  439. it.live(
  440. "BUG: without limit.input, same token count correctly triggers compaction",
  441. provideTmpdirInstance(() =>
  442. Effect.gen(function* () {
  443. const compact = yield* SessionCompaction.Service
  444. // Same model but without limit.input — uses context - output instead
  445. const model = createModel({ context: 200_000, output: 32_000 })
  446. // Same token usage as above
  447. const tokens = { input: 180_000, output: 15_000, reasoning: 0, cache: { read: 3_000, write: 0 } }
  448. // count = 198K
  449. // usable = context - output = 200K - 32K = 168K
  450. // 198K > 168K = true → compaction correctly triggered
  451. const result = yield* compact.isOverflow({ tokens, model })
  452. expect(result).toBe(true) // ← Correct: headroom is reserved
  453. }),
  454. ),
  455. )
  456. it.live(
  457. "BUG: asymmetry — limit.input model allows 30K more usage before compaction than equivalent model without it",
  458. provideTmpdirInstance(() =>
  459. Effect.gen(function* () {
  460. const compact = yield* SessionCompaction.Service
  461. // Two models with identical context/output limits, differing only in limit.input
  462. const withInputLimit = createModel({ context: 200_000, input: 200_000, output: 32_000 })
  463. const withoutInputLimit = createModel({ context: 200_000, output: 32_000 })
  464. // 170K total tokens — well above context-output (168K) but below input limit (200K)
  465. const tokens = { input: 166_000, output: 10_000, reasoning: 0, cache: { read: 5_000, write: 0 } }
  466. const withLimit = yield* compact.isOverflow({ tokens, model: withInputLimit })
  467. const withoutLimit = yield* compact.isOverflow({ tokens, model: withoutInputLimit })
  468. // Both models have identical real capacity — they should agree:
  469. expect(withLimit).toBe(true) // should compact (170K leaves no room for 32K output)
  470. expect(withoutLimit).toBe(true) // correctly compacts (170K > 168K)
  471. }),
  472. ),
  473. )
  474. it.live(
  475. "returns false when model context limit is 0",
  476. provideTmpdirInstance(() =>
  477. Effect.gen(function* () {
  478. const compact = yield* SessionCompaction.Service
  479. const model = createModel({ context: 0, output: 32_000 })
  480. const tokens = { input: 100_000, output: 10_000, reasoning: 0, cache: { read: 0, write: 0 } }
  481. expect(yield* compact.isOverflow({ tokens, model })).toBe(false)
  482. }),
  483. ),
  484. )
  485. it.live(
  486. "returns false when compaction.auto is disabled",
  487. provideTmpdirInstance(
  488. () =>
  489. Effect.gen(function* () {
  490. const compact = yield* SessionCompaction.Service
  491. const model = createModel({ context: 100_000, output: 32_000 })
  492. const tokens = { input: 75_000, output: 5_000, reasoning: 0, cache: { read: 0, write: 0 } }
  493. expect(yield* compact.isOverflow({ tokens, model })).toBe(false)
  494. }),
  495. {
  496. config: {
  497. compaction: { auto: false },
  498. },
  499. },
  500. ),
  501. )
  502. })
  503. describe("session.compaction.create", () => {
  504. it.live(
  505. "creates a compaction user message and part",
  506. provideTmpdirInstance(() =>
  507. Effect.gen(function* () {
  508. const compact = yield* SessionCompaction.Service
  509. const ssn = yield* SessionNs.Service
  510. const info = yield* ssn.create({})
  511. yield* compact.create({
  512. sessionID: info.id,
  513. agent: "build",
  514. model: ref,
  515. auto: true,
  516. overflow: true,
  517. })
  518. const msgs = yield* ssn.messages({ sessionID: info.id })
  519. expect(msgs).toHaveLength(1)
  520. expect(msgs[0].info.role).toBe("user")
  521. expect(msgs[0].parts).toHaveLength(1)
  522. expect(msgs[0].parts[0]).toMatchObject({
  523. type: "compaction",
  524. auto: true,
  525. overflow: true,
  526. })
  527. }),
  528. ),
  529. )
  530. it.live.skip(
  531. "projects a compaction message to v2 (v2 projector disabled)",
  532. provideTmpdirInstance(() =>
  533. Effect.gen(function* () {
  534. const compact = yield* SessionCompaction.Service
  535. const ssn = yield* SessionNs.Service
  536. const info = yield* ssn.create({})
  537. yield* compact.create({
  538. sessionID: info.id,
  539. agent: "build",
  540. model: ref,
  541. auto: true,
  542. overflow: true,
  543. })
  544. const v2 = yield* SessionV2.Service.use((svc) => svc.messages({ sessionID: info.id })).pipe(
  545. Effect.provide(AppNodeBuilder.build(SessionV2.node, [[SessionExecution.node, SessionExecution.noopLayer]])),
  546. )
  547. expect(v2.at(-1)).toMatchObject({
  548. type: "compaction",
  549. reason: "auto",
  550. summary: "",
  551. })
  552. }),
  553. ),
  554. )
  555. })
  556. describe("session.compaction.prune", () => {
  557. it.live(
  558. "compacts old completed tool output",
  559. provideTmpdirInstance(
  560. (dir) =>
  561. Effect.gen(function* () {
  562. const compact = yield* SessionCompaction.Service
  563. const ssn = yield* SessionNs.Service
  564. const info = yield* ssn.create({})
  565. const a = yield* ssn.updateMessage({
  566. id: MessageID.ascending(),
  567. role: "user",
  568. sessionID: info.id,
  569. agent: "build",
  570. model: ref,
  571. time: { created: Date.now() },
  572. })
  573. yield* ssn.updatePart({
  574. id: PartID.ascending(),
  575. messageID: a.id,
  576. sessionID: info.id,
  577. type: "text",
  578. text: "first",
  579. })
  580. const b: SessionV1.Assistant = {
  581. id: MessageID.ascending(),
  582. role: "assistant",
  583. sessionID: info.id,
  584. mode: "build",
  585. agent: "build",
  586. path: { cwd: dir, root: dir },
  587. cost: 0,
  588. tokens: {
  589. output: 0,
  590. input: 0,
  591. reasoning: 0,
  592. cache: { read: 0, write: 0 },
  593. },
  594. modelID: ref.modelID,
  595. providerID: ref.providerID,
  596. parentID: a.id,
  597. time: { created: Date.now() },
  598. finish: "end_turn",
  599. }
  600. yield* ssn.updateMessage(b)
  601. yield* ssn.updatePart({
  602. id: PartID.ascending(),
  603. messageID: b.id,
  604. sessionID: info.id,
  605. type: "tool",
  606. callID: crypto.randomUUID(),
  607. tool: "bash",
  608. state: {
  609. status: "completed",
  610. input: {},
  611. output: "x".repeat(200_000),
  612. title: "done",
  613. metadata: {},
  614. time: { start: Date.now(), end: Date.now() },
  615. },
  616. })
  617. for (const text of ["second", "third"]) {
  618. const msg = yield* ssn.updateMessage({
  619. id: MessageID.ascending(),
  620. role: "user",
  621. sessionID: info.id,
  622. agent: "build",
  623. model: ref,
  624. time: { created: Date.now() },
  625. })
  626. yield* ssn.updatePart({
  627. id: PartID.ascending(),
  628. messageID: msg.id,
  629. sessionID: info.id,
  630. type: "text",
  631. text,
  632. })
  633. }
  634. yield* compact.prune({ sessionID: info.id })
  635. const msgs = yield* ssn.messages({ sessionID: info.id })
  636. const part = msgs.flatMap((msg) => msg.parts).find((part) => part.type === "tool")
  637. expect(part?.type).toBe("tool")
  638. expect(part?.state.status).toBe("completed")
  639. if (part?.type === "tool" && part.state.status === "completed") {
  640. expect(part.state.time.compacted).toBeNumber()
  641. }
  642. }),
  643. {
  644. config: {
  645. compaction: { prune: true },
  646. },
  647. },
  648. ),
  649. )
  650. it.live(
  651. "skips protected skill tool output",
  652. provideTmpdirInstance((dir) =>
  653. Effect.gen(function* () {
  654. const compact = yield* SessionCompaction.Service
  655. const ssn = yield* SessionNs.Service
  656. const info = yield* ssn.create({})
  657. const a = yield* ssn.updateMessage({
  658. id: MessageID.ascending(),
  659. role: "user",
  660. sessionID: info.id,
  661. agent: "build",
  662. model: ref,
  663. time: { created: Date.now() },
  664. })
  665. yield* ssn.updatePart({
  666. id: PartID.ascending(),
  667. messageID: a.id,
  668. sessionID: info.id,
  669. type: "text",
  670. text: "first",
  671. })
  672. const b: SessionV1.Assistant = {
  673. id: MessageID.ascending(),
  674. role: "assistant",
  675. sessionID: info.id,
  676. mode: "build",
  677. agent: "build",
  678. path: { cwd: dir, root: dir },
  679. cost: 0,
  680. tokens: {
  681. output: 0,
  682. input: 0,
  683. reasoning: 0,
  684. cache: { read: 0, write: 0 },
  685. },
  686. modelID: ref.modelID,
  687. providerID: ref.providerID,
  688. parentID: a.id,
  689. time: { created: Date.now() },
  690. finish: "end_turn",
  691. }
  692. yield* ssn.updateMessage(b)
  693. yield* ssn.updatePart({
  694. id: PartID.ascending(),
  695. messageID: b.id,
  696. sessionID: info.id,
  697. type: "tool",
  698. callID: crypto.randomUUID(),
  699. tool: "skill",
  700. state: {
  701. status: "completed",
  702. input: {},
  703. output: "x".repeat(200_000),
  704. title: "done",
  705. metadata: {},
  706. time: { start: Date.now(), end: Date.now() },
  707. },
  708. })
  709. for (const text of ["second", "third"]) {
  710. const msg = yield* ssn.updateMessage({
  711. id: MessageID.ascending(),
  712. role: "user",
  713. sessionID: info.id,
  714. agent: "build",
  715. model: ref,
  716. time: { created: Date.now() },
  717. })
  718. yield* ssn.updatePart({
  719. id: PartID.ascending(),
  720. messageID: msg.id,
  721. sessionID: info.id,
  722. type: "text",
  723. text,
  724. })
  725. }
  726. yield* compact.prune({ sessionID: info.id })
  727. const msgs = yield* ssn.messages({ sessionID: info.id })
  728. const part = msgs.flatMap((msg) => msg.parts).find((part) => part.type === "tool")
  729. expect(part?.type).toBe("tool")
  730. if (part?.type === "tool" && part.state.status === "completed") {
  731. expect(part.state.time.compacted).toBeUndefined()
  732. }
  733. }),
  734. ),
  735. )
  736. })
  737. describe("session.compaction.process", () => {
  738. it.instance(
  739. "throws when parent is not a user message",
  740. Effect.gen(function* () {
  741. const test = yield* TestInstance
  742. const ssn = yield* SessionNs.Service
  743. const session = yield* ssn.create({})
  744. const msg = yield* createUserMessage(session.id, "hello")
  745. const reply = yield* createAssistantMessage(session.id, msg.id, test.directory)
  746. const msgs = yield* ssn.messages({ sessionID: session.id })
  747. const exit = yield* Effect.exit(
  748. SessionCompaction.use.process({
  749. parentID: reply.id,
  750. messages: msgs,
  751. sessionID: session.id,
  752. auto: false,
  753. }),
  754. )
  755. expect(Exit.isFailure(exit)).toBe(true)
  756. if (Exit.isFailure(exit)) {
  757. const error = Cause.squash(exit.cause)
  758. expect(error).toBeInstanceOf(Error)
  759. if (error instanceof Error) {
  760. expect(error.message).toContain(`Compaction parent must be a user message: ${reply.id}`)
  761. }
  762. }
  763. }),
  764. )
  765. it.instance(
  766. "publishes compacted event on continue",
  767. Effect.gen(function* () {
  768. const events = yield* EventV2Bridge.Service
  769. const ssn = yield* SessionNs.Service
  770. const session = yield* ssn.create({})
  771. const msg = yield* createUserMessage(session.id, "hello")
  772. const msgs = yield* ssn.messages({ sessionID: session.id })
  773. const done = yield* Deferred.make<void, Error>()
  774. const seen: string[] = []
  775. const unsub = yield* events.listen((evt) => {
  776. seen.push(evt.type)
  777. if (evt.type !== SessionCompaction.Event.Compacted.type) return Effect.void
  778. if ((evt.data as typeof SessionCompaction.Event.Compacted.data.Type).sessionID !== session.id)
  779. return Effect.void
  780. Deferred.doneUnsafe(done, Effect.void)
  781. return Effect.void
  782. })
  783. yield* Effect.addFinalizer(() => unsub)
  784. const result = yield* SessionCompaction.use.process({
  785. parentID: msg.id,
  786. messages: msgs,
  787. sessionID: session.id,
  788. auto: false,
  789. })
  790. yield* Deferred.await(done).pipe(Effect.timeout("500 millis"))
  791. expect(result).toBe("continue")
  792. expect(seen).toContain(SessionCompaction.Event.Compacted.type)
  793. expect(seen.filter((type) => type.startsWith("session.next."))).toEqual([])
  794. }),
  795. )
  796. itCompaction.instance(
  797. "marks summary message as errored on compact result",
  798. Effect.gen(function* () {
  799. const ssn = yield* SessionNs.Service
  800. const session = yield* ssn.create({})
  801. const msg = yield* createUserMessage(session.id, "hello")
  802. const msgs = yield* ssn.messages({ sessionID: session.id })
  803. const result = yield* SessionCompaction.use.process({
  804. parentID: msg.id,
  805. messages: msgs,
  806. sessionID: session.id,
  807. auto: false,
  808. })
  809. const summary = (yield* ssn.messages({ sessionID: session.id })).find(
  810. (msg) => msg.info.role === "assistant" && msg.info.summary,
  811. )
  812. expect(result).toBe("stop")
  813. expect(summary?.info.role).toBe("assistant")
  814. if (summary?.info.role === "assistant") {
  815. expect(summary.info.finish).toBe("error")
  816. expect(JSON.stringify(summary.info.error)).toContain("Session too large to compact")
  817. }
  818. }).pipe(withCompaction({ result: "compact" })),
  819. )
  820. it.instance(
  821. "adds synthetic continue prompt when auto is enabled",
  822. Effect.gen(function* () {
  823. const ssn = yield* SessionNs.Service
  824. const session = yield* ssn.create({})
  825. const msg = yield* createUserMessage(session.id, "hello")
  826. const msgs = yield* ssn.messages({ sessionID: session.id })
  827. const result = yield* SessionCompaction.use.process({
  828. parentID: msg.id,
  829. messages: msgs,
  830. sessionID: session.id,
  831. auto: true,
  832. })
  833. const all = yield* ssn.messages({ sessionID: session.id })
  834. const last = all.at(-1)
  835. expect(result).toBe("continue")
  836. expect(last?.info.role).toBe("user")
  837. expect(last?.parts[0]).toMatchObject({
  838. type: "text",
  839. synthetic: true,
  840. metadata: { compaction_continue: true },
  841. })
  842. if (last?.parts[0]?.type === "text") {
  843. expect(last.parts[0].text).toContain("Continue if you have next steps")
  844. }
  845. }),
  846. )
  847. itCompaction.instance(
  848. "persists tail_start_id for retained recent turns",
  849. Effect.gen(function* () {
  850. const ssn = yield* SessionNs.Service
  851. const session = yield* ssn.create({})
  852. yield* createUserMessage(session.id, "first")
  853. const keep = yield* createUserMessage(session.id, "second")
  854. yield* createUserMessage(session.id, "third")
  855. yield* createSummaryCompaction(session.id)
  856. const msgs = yield* ssn.messages({ sessionID: session.id })
  857. const parent = msgs.at(-1)?.info.id
  858. expect(parent).toBeTruthy()
  859. yield* SessionCompaction.use.process({
  860. parentID: parent!,
  861. messages: msgs,
  862. sessionID: session.id,
  863. auto: false,
  864. })
  865. const part = yield* readCompactionPart(session.id)
  866. expect(part?.type).toBe("compaction")
  867. expect(part?.tail_start_id).toBe(keep.id)
  868. }).pipe(withCompaction({ config: cfg({ tail_turns: 2, preserve_recent_tokens: 10_000 }) })),
  869. )
  870. itCompaction.instance(
  871. "shrinks retained tail to fit preserve token budget",
  872. Effect.gen(function* () {
  873. const ssn = yield* SessionNs.Service
  874. const session = yield* ssn.create({})
  875. yield* createUserMessage(session.id, "first")
  876. yield* createUserMessage(session.id, "x".repeat(2_000))
  877. const keep = yield* createUserMessage(session.id, "tiny")
  878. yield* createSummaryCompaction(session.id)
  879. const msgs = yield* ssn.messages({ sessionID: session.id })
  880. const parent = msgs.at(-1)?.info.id
  881. expect(parent).toBeTruthy()
  882. yield* SessionCompaction.use.process({
  883. parentID: parent!,
  884. messages: msgs,
  885. sessionID: session.id,
  886. auto: false,
  887. })
  888. const part = yield* readCompactionPart(session.id)
  889. expect(part?.type).toBe("compaction")
  890. expect(part?.tail_start_id).toBe(keep.id)
  891. }).pipe(withCompaction({ config: cfg({ tail_turns: 2, preserve_recent_tokens: 100 }) })),
  892. )
  893. itCompaction.instance(
  894. "falls back to full summary when even one recent turn exceeds preserve token budget",
  895. () => {
  896. const stub = llm()
  897. let captured = ""
  898. stub.push(reply("summary", (input) => (captured = JSON.stringify(input.messages))))
  899. return Effect.gen(function* () {
  900. const ssn = yield* SessionNs.Service
  901. const session = yield* ssn.create({})
  902. yield* createUserMessage(session.id, "first")
  903. yield* createUserMessage(session.id, "y".repeat(2_000))
  904. yield* createSummaryCompaction(session.id)
  905. const msgs = yield* ssn.messages({ sessionID: session.id })
  906. const parent = msgs.at(-1)?.info.id
  907. expect(parent).toBeTruthy()
  908. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  909. const part = yield* readCompactionPart(session.id)
  910. expect(part?.type).toBe("compaction")
  911. expect(part?.tail_start_id).toBeUndefined()
  912. expect(captured).toContain("yyyy")
  913. }).pipe(withCompaction({ llm: stub.llmLayer, config: cfg({ tail_turns: 1, preserve_recent_tokens: 20 }) }))
  914. },
  915. { git: true },
  916. )
  917. itCompaction.instance(
  918. "falls back to full summary when retained tail media exceeds preserve token budget",
  919. () => {
  920. const stub = llm()
  921. let captured = ""
  922. stub.push(reply("summary", (input) => (captured = JSON.stringify(input.messages))))
  923. return Effect.gen(function* () {
  924. const ssn = yield* SessionNs.Service
  925. const session = yield* ssn.create({})
  926. yield* createUserMessage(session.id, "older")
  927. const recent = yield* createUserMessage(session.id, "recent image turn")
  928. yield* ssn.updatePart({
  929. id: PartID.ascending(),
  930. messageID: recent.id,
  931. sessionID: session.id,
  932. type: "file",
  933. mime: "image/png",
  934. filename: "big.png",
  935. url: `data:image/png;base64,${"a".repeat(4_000)}`,
  936. })
  937. yield* createSummaryCompaction(session.id)
  938. const msgs = yield* ssn.messages({ sessionID: session.id })
  939. const parent = msgs.at(-1)?.info.id
  940. expect(parent).toBeTruthy()
  941. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  942. const part = yield* readCompactionPart(session.id)
  943. expect(part?.type).toBe("compaction")
  944. expect(part?.tail_start_id).toBeUndefined()
  945. expect(captured).toContain("recent image turn")
  946. expect(captured).toContain("Attached image/png: big.png")
  947. }).pipe(withCompaction({ llm: stub.llmLayer, config: cfg({ tail_turns: 1, preserve_recent_tokens: 100 }) }))
  948. },
  949. { git: true },
  950. )
  951. itCompaction.instance(
  952. "retains a split turn suffix when a later message fits the preserve token budget",
  953. () => {
  954. const stub = llm()
  955. let captured = ""
  956. stub.push(reply("summary", (input) => (captured = JSON.stringify(input.messages))))
  957. return Effect.gen(function* () {
  958. const test = yield* TestInstance
  959. const ssn = yield* SessionNs.Service
  960. const session = yield* ssn.create({})
  961. yield* createUserMessage(session.id, "older")
  962. const recent = yield* createUserMessage(session.id, "recent turn")
  963. const large = yield* createAssistantMessage(session.id, recent.id, test.directory)
  964. yield* ssn.updatePart({
  965. id: PartID.ascending(),
  966. messageID: large.id,
  967. sessionID: session.id,
  968. type: "text",
  969. text: "z".repeat(2_000),
  970. })
  971. const keep = yield* createAssistantMessage(session.id, recent.id, test.directory)
  972. yield* ssn.updatePart({
  973. id: PartID.ascending(),
  974. messageID: keep.id,
  975. sessionID: session.id,
  976. type: "text",
  977. text: "keep tail",
  978. })
  979. yield* createSummaryCompaction(session.id)
  980. const msgs = yield* ssn.messages({ sessionID: session.id })
  981. const parent = msgs.at(-1)?.info.id
  982. expect(parent).toBeTruthy()
  983. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  984. const part = yield* readCompactionPart(session.id)
  985. expect(part?.type).toBe("compaction")
  986. expect(part?.tail_start_id).toBe(keep.id)
  987. expect(captured).toContain("zzzz")
  988. expect(captured).not.toContain("keep tail")
  989. const filtered = MessageV2.filterCompacted(yield* MessageV2.stream(session.id))
  990. expect(filtered.map((msg) => msg.info.id).slice(0, 3)).toEqual([parent!, expect.any(String), keep.id])
  991. expect(filtered[1]?.info.role).toBe("assistant")
  992. expect(filtered[1]?.info.role === "assistant" ? filtered[1].info.summary : false).toBe(true)
  993. expect(filtered.map((msg) => msg.info.id)).not.toContain(large.id)
  994. }).pipe(withCompaction({ llm: stub.llmLayer, config: cfg({ tail_turns: 1, preserve_recent_tokens: 100 }) }))
  995. },
  996. { git: true },
  997. )
  998. itCompaction.instance(
  999. "allows plugins to disable synthetic continue prompt",
  1000. Effect.gen(function* () {
  1001. const ssn = yield* SessionNs.Service
  1002. const session = yield* ssn.create({})
  1003. const msg = yield* createUserMessage(session.id, "hello")
  1004. const msgs = yield* ssn.messages({ sessionID: session.id })
  1005. const result = yield* SessionCompaction.use.process({
  1006. parentID: msg.id,
  1007. messages: msgs,
  1008. sessionID: session.id,
  1009. auto: true,
  1010. })
  1011. const all = yield* ssn.messages({ sessionID: session.id })
  1012. const last = all.at(-1)
  1013. expect(result).toBe("continue")
  1014. expect(last?.info.role).toBe("assistant")
  1015. expect(
  1016. all.some(
  1017. (msg) =>
  1018. msg.info.role === "user" &&
  1019. msg.parts.some(
  1020. (part) => part.type === "text" && part.synthetic && part.text.includes("Continue if you have next steps"),
  1021. ),
  1022. ),
  1023. ).toBe(false)
  1024. }).pipe(withCompaction({ plugin: autocontinue(false) })),
  1025. )
  1026. it.instance(
  1027. "replays the prior user turn on overflow when earlier context exists",
  1028. Effect.gen(function* () {
  1029. const ssn = yield* SessionNs.Service
  1030. const session = yield* ssn.create({})
  1031. yield* createUserMessage(session.id, "root")
  1032. const replay = yield* createUserMessage(session.id, "image")
  1033. yield* ssn.updatePart({
  1034. id: PartID.ascending(),
  1035. messageID: replay.id,
  1036. sessionID: session.id,
  1037. type: "file",
  1038. mime: "image/png",
  1039. filename: "cat.png",
  1040. url: "https://example.com/cat.png",
  1041. })
  1042. const msg = yield* createUserMessage(session.id, "current")
  1043. const msgs = yield* ssn.messages({ sessionID: session.id })
  1044. const result = yield* SessionCompaction.use.process({
  1045. parentID: msg.id,
  1046. messages: msgs,
  1047. sessionID: session.id,
  1048. auto: true,
  1049. overflow: true,
  1050. })
  1051. const last = (yield* ssn.messages({ sessionID: session.id })).at(-1)
  1052. expect(result).toBe("continue")
  1053. expect(last?.info.role).toBe("user")
  1054. expect(last?.parts.some((part) => part.type === "file")).toBe(false)
  1055. expect(
  1056. last?.parts.some((part) => part.type === "text" && part.text.includes("Attached image/png: cat.png")),
  1057. ).toBe(true)
  1058. }),
  1059. )
  1060. it.instance(
  1061. "falls back to overflow guidance when no replayable turn exists",
  1062. Effect.gen(function* () {
  1063. const ssn = yield* SessionNs.Service
  1064. const session = yield* ssn.create({})
  1065. yield* createUserMessage(session.id, "earlier")
  1066. const msg = yield* createUserMessage(session.id, "current")
  1067. const msgs = yield* ssn.messages({ sessionID: session.id })
  1068. const result = yield* SessionCompaction.use.process({
  1069. parentID: msg.id,
  1070. messages: msgs,
  1071. sessionID: session.id,
  1072. auto: true,
  1073. overflow: true,
  1074. })
  1075. const last = (yield* ssn.messages({ sessionID: session.id })).at(-1)
  1076. expect(result).toBe("continue")
  1077. expect(last?.info.role).toBe("user")
  1078. if (last?.parts[0]?.type === "text") {
  1079. expect(last.parts[0].text).toContain("previous request exceeded the provider's size limit")
  1080. }
  1081. }),
  1082. )
  1083. itCompaction.instance(
  1084. "stops quickly when aborted during retry backoff",
  1085. () => {
  1086. const stub = llm()
  1087. stub.push(
  1088. Stream.fromAsyncIterable(
  1089. {
  1090. async *[Symbol.asyncIterator]() {
  1091. yield LLMEvent.stepStart({ index: 0 })
  1092. throw new APICallError({
  1093. message: "boom",
  1094. url: "https://example.com/v1/chat/completions",
  1095. requestBodyValues: {},
  1096. statusCode: 503,
  1097. responseHeaders: { "retry-after-ms": "10000" },
  1098. responseBody: '{"error":"boom"}',
  1099. isRetryable: true,
  1100. })
  1101. },
  1102. },
  1103. (err) => err,
  1104. ),
  1105. )
  1106. return Effect.gen(function* () {
  1107. const ssn = yield* SessionNs.Service
  1108. const events = yield* EventV2Bridge.Service
  1109. const ready = yield* Deferred.make<void>()
  1110. const session = yield* ssn.create({})
  1111. const msg = yield* createUserMessage(session.id, "hello")
  1112. const msgs = yield* ssn.messages({ sessionID: session.id })
  1113. const off = yield* events.listen((evt) => {
  1114. if (evt.type !== SessionStatus.Event.Status.type) return Effect.void
  1115. const data = evt.data as typeof SessionStatus.Event.Status.data.Type
  1116. if (data.sessionID !== session.id || data.status.type !== "retry") return Effect.void
  1117. Deferred.doneUnsafe(ready, Effect.void)
  1118. return Effect.void
  1119. })
  1120. yield* Effect.addFinalizer(() => off)
  1121. const fiber = yield* SessionCompaction.use
  1122. .process({
  1123. parentID: msg.id,
  1124. messages: msgs,
  1125. sessionID: session.id,
  1126. auto: false,
  1127. })
  1128. .pipe(Effect.forkChild)
  1129. yield* Deferred.await(ready).pipe(Effect.timeout("5 seconds"))
  1130. const start = Date.now()
  1131. yield* Fiber.interrupt(fiber)
  1132. const exit = yield* Fiber.await(fiber).pipe(Effect.timeout("250 millis"))
  1133. expect(Exit.isFailure(exit)).toBe(true)
  1134. if (Exit.isFailure(exit)) {
  1135. expect(Cause.hasInterrupts(exit.cause)).toBe(true)
  1136. expect(Date.now() - start).toBeLessThan(250)
  1137. }
  1138. }).pipe(withCompaction({ llm: stub.llmLayer }))
  1139. },
  1140. { git: true },
  1141. { timeout: 10_000 },
  1142. )
  1143. itCompaction.instance(
  1144. "does not leave a summary assistant when aborted before processor setup",
  1145. () =>
  1146. Effect.gen(function* () {
  1147. const ready = yield* Deferred.make<void>()
  1148. return yield* Effect.gen(function* () {
  1149. const ssn = yield* SessionNs.Service
  1150. const session = yield* ssn.create({})
  1151. const msg = yield* createUserMessage(session.id, "hello")
  1152. const msgs = yield* ssn.messages({ sessionID: session.id })
  1153. const fiber = yield* SessionCompaction.use
  1154. .process({
  1155. parentID: msg.id,
  1156. messages: msgs,
  1157. sessionID: session.id,
  1158. auto: false,
  1159. })
  1160. .pipe(Effect.forkChild)
  1161. yield* Deferred.await(ready).pipe(Effect.timeout("1 second"))
  1162. yield* Fiber.interrupt(fiber)
  1163. const exit = yield* Fiber.await(fiber).pipe(Effect.timeout("250 millis"))
  1164. const all = yield* ssn.messages({ sessionID: session.id })
  1165. expect(Exit.isFailure(exit)).toBe(true)
  1166. if (Exit.isFailure(exit)) expect(Cause.hasInterrupts(exit.cause)).toBe(true)
  1167. expect(all.some((msg) => msg.info.role === "assistant" && msg.info.summary)).toBe(false)
  1168. }).pipe(withCompaction({ plugin: plugin(ready) }))
  1169. }),
  1170. { git: true },
  1171. )
  1172. itCompaction.instance(
  1173. "silently drops reasoning-delta arriving without prior reasoning-start",
  1174. () => {
  1175. // Regression: PR initially auto-created a reasoning Part for orphan deltas (no preceding
  1176. // reasoning-start). Reverted to match dev — drop silently. Pinned here so any future
  1177. // change to processor.ts reasoning-delta handling triggers this test.
  1178. const stub = llm()
  1179. stub.push(
  1180. Stream.make(
  1181. LLMEvent.reasoningDelta({ id: "orphan-1", text: "stray reasoning" }),
  1182. LLMEvent.textStart({ id: "txt-0" }),
  1183. LLMEvent.textDelta({ id: "txt-0", text: "summary" }),
  1184. LLMEvent.textEnd({ id: "txt-0" }),
  1185. LLMEvent.stepFinish({ index: 0, reason: "stop", usage: basicUsage() }),
  1186. LLMEvent.finish({ reason: "stop", usage: basicUsage() }),
  1187. ),
  1188. )
  1189. return Effect.gen(function* () {
  1190. const ssn = yield* SessionNs.Service
  1191. const session = yield* ssn.create({})
  1192. const msg = yield* createUserMessage(session.id, "hello")
  1193. const msgs = yield* ssn.messages({ sessionID: session.id })
  1194. yield* SessionCompaction.use.process({
  1195. parentID: msg.id,
  1196. messages: msgs,
  1197. sessionID: session.id,
  1198. auto: false,
  1199. })
  1200. const summary = (yield* ssn.messages({ sessionID: session.id })).find(
  1201. (item) => item.info.role === "assistant" && item.info.summary,
  1202. )
  1203. expect(summary?.parts.some((part) => part.type === "reasoning")).toBe(false)
  1204. // Sanity: the text part still got through.
  1205. expect(summary?.parts.some((part) => part.type === "text" && part.text === "summary")).toBe(true)
  1206. }).pipe(withCompaction({ llm: stub.llmLayer }))
  1207. },
  1208. { git: true },
  1209. )
  1210. itCompaction.instance(
  1211. "does not allow tool calls while generating the summary",
  1212. () => {
  1213. const stub = llm()
  1214. stub.push(
  1215. Stream.make(
  1216. LLMEvent.toolCall({ id: "call-1", name: "_noop", input: {} }),
  1217. LLMEvent.stepFinish({
  1218. index: 0,
  1219. reason: "tool-calls",
  1220. usage: basicUsage(),
  1221. }),
  1222. LLMEvent.finish({
  1223. reason: "tool-calls",
  1224. usage: basicUsage(),
  1225. }),
  1226. ),
  1227. )
  1228. return Effect.gen(function* () {
  1229. const ssn = yield* SessionNs.Service
  1230. const session = yield* ssn.create({})
  1231. const msg = yield* createUserMessage(session.id, "hello")
  1232. const msgs = yield* ssn.messages({ sessionID: session.id })
  1233. yield* SessionCompaction.use.process({ parentID: msg.id, messages: msgs, sessionID: session.id, auto: false })
  1234. const summary = (yield* ssn.messages({ sessionID: session.id })).find(
  1235. (item) => item.info.role === "assistant" && item.info.summary,
  1236. )
  1237. expect(summary?.info.role).toBe("assistant")
  1238. expect(summary?.parts.some((part) => part.type === "tool")).toBe(false)
  1239. }).pipe(withCompaction({ llm: stub.llmLayer }))
  1240. },
  1241. { git: true },
  1242. )
  1243. itCompaction.instance(
  1244. "summarizes only the head while keeping recent tail out of summary input",
  1245. () => {
  1246. const stub = llm()
  1247. let captured = ""
  1248. stub.push(
  1249. reply("summary", (input) => {
  1250. captured = JSON.stringify(input.messages)
  1251. }),
  1252. )
  1253. return Effect.gen(function* () {
  1254. const ssn = yield* SessionNs.Service
  1255. const session = yield* ssn.create({})
  1256. yield* createUserMessage(session.id, "older context")
  1257. yield* createUserMessage(session.id, "keep this turn")
  1258. yield* createUserMessage(session.id, "and this one too")
  1259. yield* createCompactionMarker(session.id)
  1260. const msgs = yield* ssn.messages({ sessionID: session.id })
  1261. const parent = msgs.at(-1)?.info.id
  1262. expect(parent).toBeTruthy()
  1263. yield* SessionCompaction.use.process({
  1264. parentID: parent!,
  1265. messages: msgs,
  1266. sessionID: session.id,
  1267. auto: false,
  1268. })
  1269. expect(captured).toContain("older context")
  1270. expect(captured).not.toContain("keep this turn")
  1271. expect(captured).not.toContain("and this one too")
  1272. expect(captured).not.toContain("What did we do so far?")
  1273. }).pipe(withCompaction({ llm: stub.llmLayer }))
  1274. },
  1275. { git: true },
  1276. )
  1277. itCompaction.instance(
  1278. "anchors repeated compactions with the previous summary",
  1279. () => {
  1280. const stub = llm()
  1281. let captured = ""
  1282. stub.push(reply("summary one"))
  1283. stub.push(
  1284. reply("summary two", (input) => {
  1285. captured = JSON.stringify(input.messages)
  1286. }),
  1287. )
  1288. return Effect.gen(function* () {
  1289. const ssn = yield* SessionNs.Service
  1290. const session = yield* ssn.create({})
  1291. yield* createUserMessage(session.id, "older context")
  1292. yield* createUserMessage(session.id, "keep this turn")
  1293. yield* createCompactionMarker(session.id)
  1294. let msgs = yield* ssn.messages({ sessionID: session.id })
  1295. let parent = msgs.at(-1)?.info.id
  1296. expect(parent).toBeTruthy()
  1297. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  1298. yield* createUserMessage(session.id, "latest turn")
  1299. yield* createCompactionMarker(session.id)
  1300. msgs = MessageV2.filterCompacted(yield* MessageV2.stream(session.id))
  1301. parent = msgs.at(-1)?.info.id
  1302. expect(parent).toBeTruthy()
  1303. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  1304. expect(captured).toContain("<previous-summary>")
  1305. expect(captured).toContain("summary one")
  1306. expect(captured.match(/summary one/g)?.length).toBe(1)
  1307. expect(captured).toContain("## Important Details")
  1308. expect(captured).toContain("## Work State")
  1309. }).pipe(withCompaction({ llm: stub.llmLayer }))
  1310. },
  1311. { git: true },
  1312. )
  1313. itCompaction.instance("keeps recent pre-compaction turns across repeated compactions", () => {
  1314. const stub = llm()
  1315. stub.push(reply("summary one"))
  1316. stub.push(reply("summary two"))
  1317. return Effect.gen(function* () {
  1318. const ssn = yield* SessionNs.Service
  1319. const session = yield* ssn.create({})
  1320. const u1 = yield* createUserMessage(session.id, "one")
  1321. const u2 = yield* createUserMessage(session.id, "two")
  1322. const u3 = yield* createUserMessage(session.id, "three")
  1323. yield* createCompactionMarker(session.id)
  1324. let msgs = yield* ssn.messages({ sessionID: session.id })
  1325. let parent = msgs.at(-1)?.info.id
  1326. expect(parent).toBeTruthy()
  1327. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  1328. const u4 = yield* createUserMessage(session.id, "four")
  1329. yield* createCompactionMarker(session.id)
  1330. msgs = MessageV2.filterCompacted(yield* MessageV2.stream(session.id))
  1331. parent = msgs.at(-1)?.info.id
  1332. expect(parent).toBeTruthy()
  1333. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  1334. const filtered = MessageV2.filterCompacted(yield* MessageV2.stream(session.id))
  1335. const ids = filtered.map((msg) => msg.info.id)
  1336. expect(ids).not.toContain(u1.id)
  1337. expect(ids).not.toContain(u2.id)
  1338. expect(ids).toContain(u3.id)
  1339. expect(ids).toContain(u4.id)
  1340. expect(filtered.some((msg) => msg.info.role === "assistant" && msg.info.summary)).toBe(true)
  1341. expect(
  1342. filtered.some((msg) => msg.info.role === "user" && msg.parts.some((part) => part.type === "compaction")),
  1343. ).toBe(true)
  1344. }).pipe(withCompaction({ llm: stub.llmLayer, config: cfg({ tail_turns: 2, preserve_recent_tokens: 10_000 }) }))
  1345. })
  1346. itCompaction.instance(
  1347. "ignores previous summaries when sizing the retained tail",
  1348. Effect.gen(function* () {
  1349. const ssn = yield* SessionNs.Service
  1350. const test = yield* TestInstance
  1351. const session = yield* ssn.create({})
  1352. yield* createUserMessage(session.id, "older")
  1353. const keep = yield* createUserMessage(session.id, "keep this turn")
  1354. const keepReply = yield* createAssistantMessage(session.id, keep.id, test.directory)
  1355. yield* ssn.updatePart({
  1356. id: PartID.ascending(),
  1357. messageID: keepReply.id,
  1358. sessionID: session.id,
  1359. type: "text",
  1360. text: "keep reply",
  1361. })
  1362. yield* createCompactionMarker(session.id)
  1363. const firstCompaction = (yield* ssn.messages({ sessionID: session.id })).at(-1)?.info.id
  1364. expect(firstCompaction).toBeTruthy()
  1365. yield* createSummaryAssistantMessage(session.id, firstCompaction!, test.directory, "summary ".repeat(800))
  1366. const recent = yield* createUserMessage(session.id, "recent turn")
  1367. const recentReply = yield* createAssistantMessage(session.id, recent.id, test.directory)
  1368. yield* ssn.updatePart({
  1369. id: PartID.ascending(),
  1370. messageID: recentReply.id,
  1371. sessionID: session.id,
  1372. type: "text",
  1373. text: "recent reply",
  1374. })
  1375. yield* createCompactionMarker(session.id)
  1376. const msgs = yield* ssn.messages({ sessionID: session.id })
  1377. const parent = msgs.at(-1)?.info.id
  1378. expect(parent).toBeTruthy()
  1379. yield* SessionCompaction.use.process({ parentID: parent!, messages: msgs, sessionID: session.id, auto: false })
  1380. const part = yield* readCompactionPart(session.id)
  1381. expect(part?.type).toBe("compaction")
  1382. expect(part?.tail_start_id).toBe(keep.id)
  1383. }).pipe(withCompaction({ config: cfg({ tail_turns: 2, preserve_recent_tokens: 500 }) })),
  1384. )
  1385. })
  1386. describe("util.token.estimate", () => {
  1387. test("estimates tokens from text (4 chars per token)", () => {
  1388. const text = "x".repeat(4000)
  1389. expect(Token.estimate(text)).toBe(1000)
  1390. })
  1391. test("estimates tokens from larger text", () => {
  1392. const text = "y".repeat(20_000)
  1393. expect(Token.estimate(text)).toBe(5000)
  1394. })
  1395. test("returns 0 for empty string", () => {
  1396. expect(Token.estimate("")).toBe(0)
  1397. })
  1398. })
  1399. describe("SessionNs.getUsage", () => {
  1400. test("normalizes standard usage to token format", () => {
  1401. const model = createModel({ context: 100_000, output: 32_000 })
  1402. const result = SessionNs.getUsage({
  1403. model,
  1404. usage: usage({ inputTokens: 1000, outputTokens: 500, totalTokens: 1500 }),
  1405. })
  1406. expect(result.tokens.input).toBe(1000)
  1407. expect(result.tokens.output).toBe(500)
  1408. expect(result.tokens.reasoning).toBe(0)
  1409. expect(result.tokens.cache.read).toBe(0)
  1410. expect(result.tokens.cache.write).toBe(0)
  1411. })
  1412. test("extracts cached tokens to cache.read", () => {
  1413. const model = createModel({ context: 100_000, output: 32_000 })
  1414. const result = SessionNs.getUsage({
  1415. model,
  1416. usage: usage({ inputTokens: 1000, outputTokens: 500, totalTokens: 1500, cacheReadInputTokens: 200 }),
  1417. })
  1418. expect(result.tokens.input).toBe(800)
  1419. expect(result.tokens.cache.read).toBe(200)
  1420. })
  1421. test("handles anthropic cache write metadata", () => {
  1422. const model = createModel({ context: 100_000, output: 32_000 })
  1423. const result = SessionNs.getUsage({
  1424. model,
  1425. usage: usage({ inputTokens: 1000, outputTokens: 500, totalTokens: 1500 }),
  1426. metadata: {
  1427. anthropic: {
  1428. cacheCreationInputTokens: 300,
  1429. },
  1430. },
  1431. })
  1432. expect(result.tokens.cache.write).toBe(300)
  1433. })
  1434. test("subtracts cached tokens for anthropic provider", () => {
  1435. const model = createModel({ context: 100_000, output: 32_000 })
  1436. // AI SDK v6 normalizes inputTokens to include cached tokens for all providers
  1437. const result = SessionNs.getUsage({
  1438. model,
  1439. usage: usage({ inputTokens: 1000, outputTokens: 500, totalTokens: 1500, cacheReadInputTokens: 200 }),
  1440. metadata: {
  1441. anthropic: {},
  1442. },
  1443. })
  1444. expect(result.tokens.input).toBe(800)
  1445. expect(result.tokens.cache.read).toBe(200)
  1446. })
  1447. test("separates reasoning tokens from output tokens", () => {
  1448. const model = createModel({ context: 100_000, output: 32_000 })
  1449. const result = SessionNs.getUsage({
  1450. model,
  1451. usage: usage({ inputTokens: 1000, outputTokens: 500, reasoningTokens: 100, totalTokens: 1500 }),
  1452. })
  1453. expect(result.tokens.input).toBe(1000)
  1454. expect(result.tokens.output).toBe(400)
  1455. expect(result.tokens.reasoning).toBe(100)
  1456. expect(result.tokens.total).toBe(1500)
  1457. })
  1458. test("does not double count reasoning tokens in cost", () => {
  1459. const model = createModel({
  1460. context: 100_000,
  1461. output: 32_000,
  1462. cost: {
  1463. input: 0,
  1464. output: 15,
  1465. cache: { read: 0, write: 0 },
  1466. },
  1467. })
  1468. const result = SessionNs.getUsage({
  1469. model,
  1470. usage: usage({ inputTokens: 0, outputTokens: 1_000_000, reasoningTokens: 250_000, totalTokens: 1_000_000 }),
  1471. })
  1472. expect(result.tokens.output).toBe(750_000)
  1473. expect(result.tokens.reasoning).toBe(250_000)
  1474. expect(result.cost).toBe(15)
  1475. })
  1476. test("handles undefined optional values gracefully", () => {
  1477. const model = createModel({ context: 100_000, output: 32_000 })
  1478. const result = SessionNs.getUsage({
  1479. model,
  1480. usage: usage({ inputTokens: 0, outputTokens: 0, totalTokens: 0 }),
  1481. })
  1482. expect(result.tokens.input).toBe(0)
  1483. expect(result.tokens.output).toBe(0)
  1484. expect(result.tokens.reasoning).toBe(0)
  1485. expect(result.tokens.cache.read).toBe(0)
  1486. expect(result.tokens.cache.write).toBe(0)
  1487. expect(Number.isNaN(result.cost)).toBe(false)
  1488. })
  1489. test("calculates cost correctly", () => {
  1490. const model = createModel({
  1491. context: 100_000,
  1492. output: 32_000,
  1493. cost: {
  1494. input: 3,
  1495. output: 15,
  1496. cache: { read: 0.3, write: 3.75 },
  1497. },
  1498. })
  1499. const result = SessionNs.getUsage({
  1500. model,
  1501. usage: usage({ inputTokens: 1_000_000, outputTokens: 100_000, totalTokens: 1_100_000 }),
  1502. })
  1503. expect(result.cost).toBe(3 + 1.5)
  1504. })
  1505. test("uses authoritative Copilot billed cost when provided", () => {
  1506. const result = SessionNs.getUsage({
  1507. model: createModel({
  1508. context: 100_000,
  1509. output: 32_000,
  1510. cost: { input: 3, output: 15, cache: { read: 0.3, write: 0.3 } },
  1511. }),
  1512. usage: usage({ inputTokens: 11_774, outputTokens: 39, totalTokens: 11_813 }),
  1513. metadata: { copilot: { totalNanoAiu: 4_473_525_000 } },
  1514. })
  1515. expect(result.cost).toBe(0.04473525)
  1516. })
  1517. test("uses matching context cost tier before over-200k fallback", () => {
  1518. const model = createModel({
  1519. context: 1_000_000,
  1520. output: 32_000,
  1521. cost: {
  1522. input: 1,
  1523. output: 2,
  1524. cache: { read: 0.1, write: 0.5 },
  1525. tiers: [
  1526. {
  1527. input: 3,
  1528. output: 4,
  1529. cache: { read: 0.3, write: 1.5 },
  1530. tier: { type: "context", size: 200_000 },
  1531. },
  1532. {
  1533. input: 5,
  1534. output: 6,
  1535. cache: { read: 0.5, write: 2.5 },
  1536. tier: { type: "context", size: 500_000 },
  1537. },
  1538. ],
  1539. experimentalOver200K: {
  1540. input: 100,
  1541. output: 100,
  1542. cache: { read: 100, write: 100 },
  1543. },
  1544. },
  1545. })
  1546. const result = SessionNs.getUsage({
  1547. model,
  1548. usage: usage({
  1549. inputTokens: 650_000,
  1550. outputTokens: 100_000,
  1551. totalTokens: 750_000,
  1552. cacheReadInputTokens: 100_000,
  1553. }),
  1554. })
  1555. expect(result.tokens.input).toBe(550_000)
  1556. expect(result.cost).toBe(2.75 + 0.6 + 0.05)
  1557. })
  1558. test("falls back to over-200k pricing when no cost tier matches", () => {
  1559. const model = createModel({
  1560. context: 1_000_000,
  1561. output: 32_000,
  1562. cost: {
  1563. input: 1,
  1564. output: 2,
  1565. cache: { read: 0.1, write: 0.5 },
  1566. tiers: [
  1567. {
  1568. input: 5,
  1569. output: 6,
  1570. cache: { read: 0.5, write: 2.5 },
  1571. tier: { type: "context", size: 500_000 },
  1572. },
  1573. ],
  1574. experimentalOver200K: {
  1575. input: 3,
  1576. output: 4,
  1577. cache: { read: 0.3, write: 1.5 },
  1578. },
  1579. },
  1580. })
  1581. const result = SessionNs.getUsage({
  1582. model,
  1583. usage: usage({ inputTokens: 300_000, outputTokens: 100_000, totalTokens: 400_000 }),
  1584. })
  1585. expect(result.cost).toBe(0.9 + 0.4)
  1586. })
  1587. test.each(["@ai-sdk/anthropic", "@ai-sdk/amazon-bedrock", "@ai-sdk/google-vertex/anthropic"])(
  1588. "computes total from components for %s models",
  1589. (npm) => {
  1590. const model = createModel({ context: 100_000, output: 32_000, npm })
  1591. // AI SDK v6: inputTokens includes cached tokens for all providers
  1592. const item = usage({
  1593. inputTokens: 1000,
  1594. outputTokens: 500,
  1595. totalTokens: 1500,
  1596. cacheReadInputTokens: 200,
  1597. })
  1598. if (npm === "@ai-sdk/amazon-bedrock") {
  1599. const result = SessionNs.getUsage({
  1600. model,
  1601. usage: item,
  1602. metadata: {
  1603. bedrock: {
  1604. usage: {
  1605. cacheWriteInputTokens: 300,
  1606. },
  1607. },
  1608. },
  1609. })
  1610. // inputTokens (1000) includes cache, so adjusted = 1000 - 200 - 300 = 500
  1611. expect(result.tokens.input).toBe(500)
  1612. expect(result.tokens.cache.read).toBe(200)
  1613. expect(result.tokens.cache.write).toBe(300)
  1614. // total = adjusted (500) + output (500) + cacheRead (200) + cacheWrite (300)
  1615. expect(result.tokens.total).toBe(1500)
  1616. return
  1617. }
  1618. const result = SessionNs.getUsage({
  1619. model,
  1620. usage: item,
  1621. metadata: {
  1622. anthropic: {
  1623. cacheCreationInputTokens: 300,
  1624. },
  1625. },
  1626. })
  1627. // inputTokens (1000) includes cache, so adjusted = 1000 - 200 - 300 = 500
  1628. expect(result.tokens.input).toBe(500)
  1629. expect(result.tokens.cache.read).toBe(200)
  1630. expect(result.tokens.cache.write).toBe(300)
  1631. // total = adjusted (500) + output (500) + cacheRead (200) + cacheWrite (300)
  1632. expect(result.tokens.total).toBe(1500)
  1633. },
  1634. )
  1635. test("extracts cache write tokens from vertex metadata key", () => {
  1636. const model = createModel({ context: 100_000, output: 32_000, npm: "@ai-sdk/google-vertex/anthropic" })
  1637. const result = SessionNs.getUsage({
  1638. model,
  1639. usage: usage({ inputTokens: 1000, outputTokens: 500, totalTokens: 1500, cacheReadInputTokens: 200 }),
  1640. metadata: {
  1641. vertex: {
  1642. cacheCreationInputTokens: 300,
  1643. },
  1644. },
  1645. })
  1646. expect(result.tokens.input).toBe(500)
  1647. expect(result.tokens.cache.read).toBe(200)
  1648. expect(result.tokens.cache.write).toBe(300)
  1649. })
  1650. })