diff --git a/PERF.md b/PERF.md index c17ee82145..c3a943c7d5 100644 --- a/PERF.md +++ b/PERF.md @@ -6,6 +6,44 @@ two thirds off the embedded corpora. This file is what keeps it. Every win below is defended by something that goes red locally, in `go test` or in `make check`, with a message that says what happened. +## Context recovery bounds + +Conversation request admission sums the existing encoded messages and tool schemas; +it performs no tokenizer call, network lookup or extra model request. Image payload +bytes are replaced by a token allowance. The margin is 5% of the effective endpoint +window, bounded to 512–8,192 tokens. An unspecified output allowance is bounded by +one quarter of the window and the configured completion reserve; shrinking it retains +up to 512 tokens as the useful minimum (one eighth for very small windows). A thinking +budget shrinks to the room left beside an answer of up to 1,024 tokens (a sixteenth of +the window) and is dropped below 1,024, so thinking never refuses a request. Endpoints +that take no tools are left out of the window a tool-carrying request is measured +against. + +Manual and emergency reductions retain 4,096 recent tokens, capped to an eighth of +the window, and always retain the latest assistant/tool batch. Recovery is bounded +to two changed-request attempts per failed generation and resets after a successful +response. Tests assert request counts, fitting budgets, tool pairing and actions that +execute exactly once; they do not wait on real clocks. + +Automatic conversation profiles use the existing 32,000-token threshold at request +boundaries. Only crossing the threshold rebuilds the default prompt and belt; unchanged +profiles do no schema work. Explicitly loaded capabilities survive the rebuild. Automatic +no-op compaction emits no seam events, while manual commands retain their no-op feedback. + +A compaction pass makes **one model call only when its free rungs fail**: a summary +(`internal/session/compact_summary.go`) is written when stubbing and folding leave the +transcript above the pass's line, and never when the tool definitions alone exceed that +line. An automatic or manual pass needs at least **1,024 tokens** of region the previous +summary has not read; a refused request's recovery needs **128**. It keeps the **three** +most recent person messages and everything after them when that still reaches the line, +then two, then the latest with the reply before it, then the latest alone. A summary is +asked for at most half its region and at most a twentieth of the window (**128–4,096** +tokens). A refusal that states its figures reclaims what is missing plus a thirty-second +of the window; one that does not reclaims a quarter of the transcript. A region larger than one request is summarized in chunks sized to the window, +each bounded to **two minutes**; the session lock is released during every call. The +render the summarizer reads caps a tool result at 2,000 bytes and a call's arguments at +400. Tests use scripted completers and assert request counts and sizes, not clocks. + ## Connection recovery bounds `internal/provider/connectivity.go` limits a connection-recovery episode to @@ -1537,10 +1575,10 @@ otherwise getting 57–61% of its prompt back from. So the ceiling is a MEASUREMENT now and not a constant: the narrowest prompt this model has actually been refused for being too long, learned from the overflow refusal itself and remembered across processes -(`internal/provider`'s `NoteServedWindow` / `ServedWindow`, applied by -`session.TrustedWindowFor`). A model nobody has refused is believed; one that has -refused is capped at what it refused, for good. The 386k incident now costs one -turn per model per machine instead of every model for ever. +(`internal/provider`'s `NoteServedWindow` / `ServedWindow`, applied while the provider +sizes each assembled request). `session.TrustedWindowFor` now returns the catalog window +unchanged: a model-only memo cannot say which endpoint can serve a request with tools. +The provider uses the refused endpoint's evidence when that request is sized. Two guards stand behind that trade and neither is new: `guardOversizeRequest` still shrinks a transcript that has grown past the trusted window before it goes diff --git a/cmd/codeaf/chatv3.go b/cmd/codeaf/chatv3.go index 36191f6d41..7b4ff2a8e6 100644 --- a/cmd/codeaf/chatv3.go +++ b/cmd/codeaf/chatv3.go @@ -1031,7 +1031,14 @@ func openV3Launch(proc *v3Process, opts v3Options) (*v3Launch, error) { // --reasoning went to every model blind — and on a router, a knob no // endpoint publishes is not a 400 but a 404 with no endpoints left to // serve the request (internal/provider's endpoints.go). - SupportsParameter: activeModels.SupportsParameter, + // + // AND IT IS ASKED ABOUT THE ID THE SERVICE IS SENT. A model behind a + // connected service is named here with the service's written prefix + // (`stub/z-ai/glm-5.3-flash`), and the adapter asks the same catalog + // about the id it puts on the wire (`z-ai/glm-5.3-flash`): answered only + // about the prefixed name, the session kept the working page for a + // model the adapter was already sending no tools (chatpage.go). + SupportsParameter: v3SupportsParameter(proc.Shelf, activeModels), ReasoningProfile: config.ReasoningProfileSeam(activeModels), // And the model's own published price, which is what bounds the latency // ask: this session wants the fastest endpoint, not the dearest one @@ -2492,6 +2499,25 @@ func v3AnswersText(outputs []string) bool { // The file is read at most once per session: it is the same rows for the whole // warming window, and re-reading it per message would put I/O on the message // path to learn nothing new. +// v3SupportsParameter is the catalog's answer about a model, asked first as +// named and then as the id its service is sent — the same id the adapter asks +// about, so the session's page and the adapter's body cannot disagree about +// what the model accepts. +func v3SupportsParameter(shelf *v3ModelShelf, models *catalog.Catalog) func(string, string) (bool, bool) { + return func(model, parameter string) (bool, bool) { + if models == nil { + return false, false + } + if supported, known := models.SupportsParameter(model, parameter); known { + return supported, known + } + if bare := shelf.wireModel(model); bare != "" && !strings.EqualFold(bare, model) { + return models.SupportsParameter(bare, parameter) + } + return false, false + } +} + func v3SeesImages(models v3Catalog) func(string) bool { var once sync.Once var cached []tui3.Model diff --git a/cmd/codeaf/chatv3_modelshelf.go b/cmd/codeaf/chatv3_modelshelf.go index 7db5277755..07d763b625 100644 --- a/cmd/codeaf/chatv3_modelshelf.go +++ b/cmd/codeaf/chatv3_modelshelf.go @@ -173,6 +173,22 @@ func (s *v3ModelShelf) contextWindow(model string) int { return v3ContextWindow(s.modelsForService(service), bare) } +// wireModel is the id a model is sent to its service under: the service's +// written prefix and any thinking level taken off. A shelf that knows no +// services answers the model as named. +func (s *v3ModelShelf) wireModel(model string) string { + model, _ = roles.SplitEffort(strings.TrimSpace(model)) + if s == nil { + return model + } + sources := s.sourcesNow() + if sources.Empty() { + return model + } + _, bare := sources.For(model) + return bare +} + // sourcesNow is the service set the shelf was last aligned with. func (s *v3ModelShelf) sourcesNow() modelsource.Set { s.mu.RLock() diff --git a/docs/changes/unreleased/1658-compaction-summaries-and-recovery.md b/docs/changes/unreleased/1658-compaction-summaries-and-recovery.md new file mode 100644 index 0000000000..448ba24748 --- /dev/null +++ b/docs/changes/unreleased/1658-compaction-summaries-and-recovery.md @@ -0,0 +1,32 @@ +--- +kind: changed +title: compaction can end with a summary, refused requests recover, and a summary leaves a visible line +pr: 1658 +surface: [chat, engine, remote] +invalidates: + - "Compaction never summarized: it only stubbed old tool results and folded old assistant work. When those are not enough, the conversation's own model now summarizes the oldest part, older user messages included. An automatic or manual pass keeps the three most recent user messages when no cut can reach its line; recovery from a refused request can keep fewer to reclaim room. /compact always goes on to a summary in the same pass when there is enough older material." + - "A request's window was the smallest among all of a model's endpoints. A request that carries tools is now measured only against endpoints that take tools. Default routing sends no require_parameters, so the router can still hand it to one that takes none: a size refusal from such an endpoint is sent once more before anything is shortened, its window is not used for later tool requests unless the request pins that endpoint, and every learned endpoint window lapses after 30 minutes." + - "The lane sheet read tool support from supports_tool_choice.function. It reads supported_parameters, the list the router filters on under require_parameters." + - "The xhigh and max thinking budgets were reserved in full whatever the window. They now shrink to the room the prompt leaves and are dropped below 1,024 tokens." + - "A model that takes no tools was sent the tool list and retried with `Retry 1/1: removed tools`. It is sent none from the start, reads a short chat page, and the person is told once why. A model the catalog does not know is sent no tools for 30 minutes only after a retry without them was answered." + - "/compact waited ten seconds and then said `compact failed: the engine did not answer in time` about a pass that went on to land. A new surface waits up to five minutes (session.CompactPatience), and a message sent meanwhile is not queued behind it. An old surface talking to a new engine still reports that ten-second failure even though the pass can land later: the wire Version stays at 20 because no frame or method changed, and Decision 3 in docs/REMOTE.md reserves a version bump for a protocol change." + - "MethodCompact rode the remote ordered lane. It is its own call class, classWork, on a goroutine of its own." + - "A /compact that changed nothing said only `nothing to compact`. It now says why." + - "After /compact the status line kept the last request's weight until the next message was sent. It drops as soon as the pass lands." + - "The post-compaction meter and the ⚭ line counted only transcript text while the next request still carried tool definitions. Both now include the sent definitions in their estimates." + - "A switch back to a model without tools carried earlier tool-call protocol messages and could be refused. Its request now carries the earlier calls and results as readable text; tool-capable requests keep their original history bytes." + - "DeepInfra's `Requested input length … exceeds maximum input length …` refusal was retried as an ordinary error. It now enters overflow recovery and teaches the endpoint's stated limit." + - "A finished compaction was a full-width rule that folded into `▸ worked` when the turn ended. It is one dim `⚭ compacted` line. A pass that wrote a summary stands outside the fold after the turn, decided by the new session.Event.Summarized count; a pass that only folded or stubbed still folds with the turn's work, as #1627 intended." + - "A /compact pass that folded work while its summary failed said only `compacted`. It now says `summary skipped: ` in its note, including over a remote engine; a successful remote reply to an older surface still reads as plain success. A visit to Home during the pass no longer loses that conversation's note or meter update." + - "An unobserved helper call could consume a tool-less model's one visible notice. The notice is now spent only on a call with a stream observer." + - "A strict endpoint pin could resend the same oversized tool request to that endpoint and ignore its learned limit on later tool requests. It now pays one refusal and sizes the next request against the pinned endpoint's limit. A routed resend logs its first 400 as well as the answer. Equal limits refresh their disk date, and future dates are clamped." + - "An ordinary request could send `max_tokens` even when its computed output ceiling did not bind. It now omits that field until the ceiling actually binds." + - "A summary on a model whose lowest listed thinking effort was high could spend its whole output cap thinking and return empty. The effort word still travels as asked, but the provider now sizes the thinking room for the lowest level the model lists (a model that lists only high and xhigh runs at high at the least); an empty length finish gets one retry at twice the answer allowance, and both attempts are counted." + - "When no cut could reach the line and only two person messages remained after a summary, /compact could summarize away one of them to reach the minimum worth a call. Ordinary and manual passes now keep every recent person message the three-message ladder protects; only refusal recovery may keep fewer." + - "A rolling summary could omit specific facts from the previous summary. Its instruction now explicitly carries those facts forward word for word unless newer conversation supersedes them." +--- + +A conversation could fill its window and then be stuck: the request was refused +as too long and `/compact` found nothing to do. The summary is the rung that was +missing, and the size check now counts only what a request really carries, so a +refusal names a shortfall that compaction can actually close. diff --git a/docs/design/icons/DESIGN.md b/docs/design/icons/DESIGN.md index cb1082b822..8a9eb22a2b 100644 --- a/docs/design/icons/DESIGN.md +++ b/docs/design/icons/DESIGN.md @@ -127,6 +127,15 @@ its verdict. The dim action mark is distinct from the failed-work mark. It is drawn only where a click can remove the attachment, and is resolved explicitly through the vocabulary. +### Conversation compaction + +| Meaning | Slot | Plain | Nerd font | ASCII | +| --- | --- | --- | --- | --- | +| working context shortened | `GCompacted` | `⚭` | nf-fa-compress | `#` | + +The Font Awesome 4 `fa-compress` mark is U+F066. Both the automatic pass line and +the `/compact` reply resolve this slot through the chat surface's glyph door. + ### File kinds (chips, and the gutter beside a call that made or opened one) | Kind | Slot | Plain | Nerd font | ASCII | diff --git a/internal/iconlaw/iconlaw_test.go b/internal/iconlaw/iconlaw_test.go index bf45f767cd..e4bdb4be9e 100644 --- a/internal/iconlaw/iconlaw_test.go +++ b/internal/iconlaw/iconlaw_test.go @@ -74,6 +74,7 @@ var ownedRunes = map[rune]string{ '⇉': "tokens.GActionCoordinate", '◷': "tokens.GActionWait", '▪': "tokens.GActionWork", + '⚭': "tokens.GCompacted", '⌕': "tokens.GSearch", '✎': "tokens.GWrite", '⌾': "tokens.GFileImage", diff --git a/internal/lane/lanestub/lanestub.go b/internal/lane/lanestub/lanestub.go index 769adda32a..5b7f033aa7 100644 --- a/internal/lane/lanestub/lanestub.go +++ b/internal/lane/lanestub/lanestub.go @@ -675,6 +675,7 @@ type sheetEndpoint struct { Quantization string `json:"quantization"` ContextLength int `json:"context_length"` MaxCompletionTokens int `json:"max_completion_tokens"` + SupportedParameters []string `json:"supported_parameters"` Pricing sheetPricing `json:"pricing"` SupportsToolChoice sheetToolChoice `json:"supports_tool_choice"` Status int `json:"status"` @@ -725,6 +726,16 @@ func (s *Server) serveSheet(w http.ResponseWriter, r *http.Request) { writeJSON(w, http.StatusOK, body) } +// parameters is the list a real router filters on under `require_parameters`, +// and the one the sheet reads tool support from: "tools" is on it exactly +// when the lane says it takes them. +func (l Lane) parameters() []string { + if l.Tools { + return []string{"max_tokens", "reasoning", "tools", "tool_choice"} + } + return []string{"max_tokens", "reasoning"} +} + // row is the sheet's account of one lane. Percentiles a profile did not state // are derived from what it actually does, with the spread a real lane has — // which makes "the sheet is right about this lane" the default and leaves @@ -749,6 +760,7 @@ func (l Lane) row(model string) sheetEndpoint { Quantization: l.Quant, ContextLength: l.Context, MaxCompletionTokens: l.MaxOut, + SupportedParameters: l.parameters(), Pricing: sheetPricing{ Prompt: money(l.PriceIn), Completion: money(l.PriceOut), diff --git a/internal/lane/sheet.go b/internal/lane/sheet.go index 761a187690..024410306a 100644 --- a/internal/lane/sheet.go +++ b/internal/lane/sheet.go @@ -8,6 +8,7 @@ import ( "net/url" "os" "path/filepath" + "slices" "sort" "strconv" "strings" @@ -1171,7 +1172,15 @@ type wireEndpoint struct { Completion string `json:"completion"` InputCacheRead string `json:"input_cache_read"` } `json:"pricing"` - SupportsToolChoice struct { + // SupportedParameters is the list the router filters on when a request + // says `require_parameters`, which every request from this program does — + // so it is the one answer to "will a request carrying tools reach this + // lane". SupportsToolChoice describes which tool_choice VALUES the lane + // takes, a different question: on 2026-09-28 deepseek-v3.2's sheet had + // GMICloud, AtlasCloud and Alibaba taking tools with no forced-function + // choice, and Mara offering the choice while taking no tools at all. + SupportedParameters []string `json:"supported_parameters"` + SupportsToolChoice struct { Function bool `json:"function"` } `json:"supports_tool_choice"` // Status is the router's own health word for the endpoint: zero is healthy, @@ -1183,6 +1192,16 @@ type wireEndpoint struct { ThroughputLast30m wirePercentiles `json:"throughput_last_30m"` } +// takesTools reads a lane's tool support from the parameter list the router +// itself filters on, and falls back to the tool_choice block only for a sheet +// that publishes no list, which is what an older router or a stub sends. +func (item wireEndpoint) takesTools() bool { + if item.SupportedParameters == nil { + return item.SupportsToolChoice.Function + } + return slices.Contains(item.SupportedParameters, "tools") +} + // decodeSheet reads the endpoints body one row at a time. The error it returns // is about the envelope; a row it could not read is skipped in silence, which // is the whole point of decoding this way. @@ -1212,7 +1231,7 @@ func decodeSheet(model string, body io.Reader) ([]Row, map[ID]string, error) { rows = append(rows, Row{ ID: id, Facts: Facts{ - Tools: item.SupportsToolChoice.Function, + Tools: item.takesTools(), Quant: strings.TrimSpace(item.Quantization), MaxOut: item.MaxCompletionTokens, Context: item.ContextLength, diff --git a/internal/lane/sheet_test.go b/internal/lane/sheet_test.go index 9b730e834a..ad2d66b461 100644 --- a/internal/lane/sheet_test.go +++ b/internal/lane/sheet_test.go @@ -465,3 +465,30 @@ func TestTheBeatPrimesTheBeliefFromEveryReading(t *testing.T) { } } } + +// TOOL SUPPORT IS READ FROM THE LIST THE ROUTER FILTERS ON. Every request from +// this program says `require_parameters`, so a lane whose supported_parameters +// lacks "tools" never receives a request that carries them, whatever its +// tool_choice block says. The two rows are deepseek-v3.2's GMICloud and Mara as +// the router published them on 2026-09-28, where the two fields disagree in +// both directions; the third row publishes no list and keeps the old reading. +func TestToolSupportIsReadFromTheParameterListTheRouterFiltersOn(t *testing.T) { + body := `{"data":{"endpoints":[ + {"provider_name":"GMICloud","context_length":163840,"supported_parameters":["max_tokens","tools","tool_choice"],"supports_tool_choice":{"function":false,"auto":true}}, + {"provider_name":"Mara","context_length":32768,"supported_parameters":["max_tokens","temperature"],"supports_tool_choice":{"function":true,"auto":true}}, + {"provider_name":"Listless","context_length":65536,"supports_tool_choice":{"function":true}} + ]}}` + rows, _, err := decodeSheet("deepseek/deepseek-v3.2", strings.NewReader(body)) + if err != nil { + t.Fatal(err) + } + want := map[string]bool{"GMICloud": true, "Mara": false, "Listless": true} + for _, row := range rows { + if row.Facts.Tools != want[row.ID.Lane] { + t.Errorf("%s reads Tools=%v, want %v", row.ID.Lane, row.Facts.Tools, want[row.ID.Lane]) + } + } + if len(rows) != len(want) { + t.Fatalf("decoded %d rows, want %d", len(rows), len(want)) + } +} diff --git a/internal/manual/chat/commands.md b/internal/manual/chat/commands.md index 506f8c5384..5ac8fd0f83 100644 --- a/internal/manual/chat/commands.md +++ b/internal/manual/chat/commands.md @@ -192,7 +192,7 @@ Canonical word, the other words it answers to, its argument form, and what it do | `/new` | `/clear`, `/clean`, `/reset` | — | closes this session and starts a fresh one | | `/drafts` | — | — | lists cleared drafts, newest first; enter restores one to the box and `d` lets one go; an empty ring says `no cleared draft is waiting` | | `/resume` | `/sessions` | — | opens the earlier-conversations picker | -| `/compact` | — | — | summarizes the conversation now | +| `/compact` | — | — | shortens the conversation now | ## Home, project and file context commands — what does /workspace path do @@ -426,30 +426,37 @@ new session failed: A close that fails says so and the surface continues. A launcher that cannot build the replacement says so, and nothing is replaced. -## /compact — summarize the conversation now - -`/compact` notes `compacting…` immediately and runs the compaction off the loop. - -**There is no success message.** A compaction that worked is silent — the note that it -started is all you get. - -A failure comes back as: - -``` -compact failed: -``` - +## /compact — shorten the conversation now + +`/compact` notes `compacting…` immediately and reduces older completed work, even below +the automatic threshold. It first turns old tool results into pointers and folds older +assistant work, which costs nothing. If enough older conversation remains, the conversation's +own model writes a **summary** in the same pass, replacing older messages. +Your three most recent messages and +everything after them stay word for word unless a cut that keeps fewer is what reaches the +line: codeaf chooses the cut that keeps the most and still gets under it — three, two, your +latest with the reply before it, or your latest alone. When no cut can reach the line, +`/compact` keeps all of your last three messages that exist, since summarizing more +would not reach it either. With only two since the last summary, it keeps both; if +there is too little before them, it writes no summary and says why. Only recovery +from a refused request then keeps fewer. The +latest message and the system prompt always stay word for word. The full record +stays in the session journal, and the summary names that file. + +Success reports `⚭ compacted · about N to M tokens` when the measured count fell, or +`⚭ compacted` without a size when it did not; the figures are estimates, and the line +stays in the conversation as the answer to your command. When nothing +changed it says `nothing to compact — ` and why: for example `only ~400 tokens since the +last summary — too little to summarize`, `there is nothing before your last 3 messages to +summarize`, or `the model could not write a summary: ` and the reason. The status line's +count drops as soon as the pass lands. +Other failures say `compact failed: ` followed by the reason. The pass runs off the input +loop, so the surface stays responsive, and a message you send while it runs is not held +behind it; if that message is too long to send before the pass lands, it waits for the pass +and then goes. A summary on a slow model can take a minute; the chat waits up to five. If the +engine is still working after that, it says `still compacting — it is taking longer than usual and finishes on its own; the token count in the status line drops when it lands`. `/compact` has no argument form and no alias. -**A compaction costs nothing and asks no model.** It is two mechanical passes over -the messages this session already has: tool results the model has already used -become pointers to their own bytes, and if that is not enough the oldest assistant -work is replaced by one marker line naming how much went and where it can be read. -Your own words are never folded. There is **no summariser** and there is **no -`compaction` role in settings** — there was one, and it was a priced row wired to -nothing. What the model is handed instead of a summary is the state card, which is -maintained a little at a time by the reader that runs after each turn. - ## /rewind — go back to an earlier point in the conversation `/rewind` (or `/undo`, `/back`) opens the **rewind timeline**: a fullscreen list of the diff --git a/internal/manual/chat/compacting-over-and-over.md b/internal/manual/chat/compacting-over-and-over.md index 7c598106d2..afe9e1f4ca 100644 --- a/internal/manual/chat/compacting-over-and-over.md +++ b/internal/manual/chat/compacting-over-and-over.md @@ -40,8 +40,8 @@ you have not set one yourself; the next section is how to set one. Two things used to make a big model fold like a small one, and both are fixed: - codeaf refused to believe any claim above 256,000 tokens, for every model alike. That - ceiling is gone; what can lower a claim now is an endpoint actually **refusing** a request - for being too long, which codeaf writes down and never trusts that model past again. + ceiling is gone; request admission now reads endpoint-specific windows and explicit limits reported by + refusals. A rejected prompt size is not saved as a model-wide window. - work that left the conversation — a task's worker, an adaptive run's worker, the reader that checks a task — was handed nothing at all when its model differed from yours, and so folded against the conservative 128,000-token default whatever its own model held. @@ -105,8 +105,9 @@ keep appearing is the sign of the defect this page describes. ## What happens to tool results while one long answer is still working A running answer keeps its assistant notes, tool calls, their exact arguments and every -mutating result in the model's context. General conversation compaction does not fold that -current-turn work. This is deliberate: it is the working record of what the model tried +mutating result in the model's context. Routine conversation compaction does not fold that +current-turn work. Manual compaction and necessary overflow recovery may archive older +completed batches while retaining the newest batch and a smaller recent working tail. This is deliberate: it is the working record of what the model tried and what it changed, and removing it can make the model inspect the same files or repeat an edit. @@ -124,15 +125,19 @@ the newest slice stays whole. Each older slice becomes a pointer naming the file `offset` and `limit` that read it, and no copy is filed — the file itself is where those bytes came from, and `read` brings the slice back. -## A pass that cannot reach its target +## A pass that cannot reach its target — when folding is not enough The fold walks the oldest assistant work first and stops at the target, but it never folds your own messages, the system prompt, the verbatim tail, or a tool call whose result has been stubbed. A conversation that is mostly your own words and recent work can run out of -foldable material above the target. The pass still succeeds with what it took, and the -`compacted · …` line reports the real counts; the next step's check may then fire again, -honestly, because there was nothing more to take. That is the one case where two passes -in quick succession are not a defect. +foldable material above the target. + +When that leaves the conversation above the line that fired the pass, the pass ends with a +**summary**: the conversation's own model rewrites the oldest part of the conversation as +one note, and the `compacted · …` line says `summarized N messages`. See *When compaction +writes a summary*. A summary is skipped when the tool definitions alone already exceed the +line — no summary could get under it — and then the next step may fire again, honestly, +because there was nothing more to take. ## What happened to the earlier messages — where did the folded messages go — how do I get the compacted text back, why it loses the earlier part of our chat @@ -155,3 +160,91 @@ ever the journal. Neither invents a file to open. Scrolling up above the fold on the screen also still shows the words; what shrank is the model's copy, not yours. The fold is not unrecoverable. + +## Why /compact says nothing to compact — manual compaction before the automatic trigger + +`/compact` now has its own reduction policy. It can fold older completed assistant work +before the automatic trigger, including completed batches inside one long turn. It keeps +your messages, the system prompt, the newest assistant/tool batch, and 4,096 recent tokens +(at most an eighth of the trusted window). Then, in the same pass, it summarizes whatever +older conversation is left (see *When compaction writes a summary*), so one `/compact` goes +as far as it can; a second one right after has nothing left to do. + +The old command reused the automatic target. A conversation with 60,000 tokens on a 128k +model could have older history and still receive `session: nothing to compact`, because it +was below that target. The manual command no longer has that threshold gate. + +A no-op says `nothing to compact — ` and why: + +- `only ~400 tokens since the last summary — too little to summarize` (or `before your + latest message`): a summary needs about 1,000 tokens of conversation it has not read; +- `nothing new since the last summary`, or `there is nothing before your last 3 messages + to summarize` (or `your latest message`): the messages kept word for word are all that + is left; +- `the model could not write a summary: …`, `the model declined to write a summary`, + `the model's summary came back empty or unreadable`, `the summary was interrupted`, or + `the conversation changed while the summary was being written` — the conversation is + left exactly as it was. + +When the free steps did shorten something but the summary did not land, the pass still +reports what it folded. Its line includes `summary skipped: ` and why the summary did not +land — the model could not write it, declined, sent back nothing usable, was interrupted, +or the conversation changed meanwhile; +an automatic line may have size and journal details after that clause. `/compact` says +`⚭ compacted · about X to Y tokens · summary skipped: ` when its size fell, or +`⚭ compacted · summary skipped: ` when there was no size drop. + +None of these means the next request fits: admission also counts schemas, replayed +reasoning and reserved output. + +## Why the provider says maximum context length when the status shows 20 percent — a request refused as too long + +The status shows the model catalog's window. The serving endpoint may have less room, and +its window must hold input **plus output**, including thinking. codeaf budgets the assembled +request against known endpoint limits before sending and remembers explicit limits from +errors by base URL, model and endpoint. Old rejected-prompt-size guesses are ignored. + +Recovery shortens by what is missing plus a little room — a thirty-second of the window — +rather than a quarter of the conversation, and it will summarize even a small older part +when that is what the request is short of. Overflow recovery can run twice per failed +generation, only while the request changes. +A DeepInfra refusal saying `Requested input length … exceeds maximum input length …` +counts as an overflow; its stated maximum is used as that endpoint's limit. +A successful response resets the allowance. A second overflow later in a long tool turn +can therefore recover instead of ending the turn just because it compacted earlier. +If protected material still cannot fit, codeaf explains that locally; it does not +knowingly send the same oversized request again. + +## When compaction writes a summary — does codeaf summarize my conversation, what the summary keeps + +Yes, as a last resort. Folding and pointers are tried first because they are free and +nothing is paraphrased. A summary is written only when they cannot bring the conversation +under the line the pass needs: the automatic threshold, the size a refused request has to +shrink to. `/compact` is the exception: it always goes on to summarize whatever older +conversation is left, because you asked for it as short as it can be. + +The summary is written by **the model you are talking to**, with no tools, and it is billed +like any other call, counting toward the session's spending. It replaces the +oldest part of the conversation — your older messages and the assistant's work alike — +with one note that starts `[context compacted]`. It never touches: + +- the system prompt; +- your **up to three most recent messages** and everything after them. If a cut can reach + the line, codeaf keeps as many of those messages as that cut allows. If no cut can + reach it, an automatic pass or `/compact` still keeps all of the most recent three + that exist: summarizing more would not reach the line either. With only two messages + since the last summary, `/compact` leaves both word for word when neither can reach + the line; if there is too little older conversation, it writes no summary and says why. + Recovery from a refused request tries two, then + your latest message **with the reply just before it** (what "translate it" or + "keep going" is about), then your latest alone — the message being answered always stays; +- the turn that is running. + +Messages codeaf writes into the conversation itself — the note after a reply was cut off +at the output limit, a `[carry on]` — are not counted as yours. + +A later summary folds the earlier one in, so there is only ever one note. The original +words stay in the session journal, which the note names, and on your screen when you +scroll up. A summary that fails, is not prose, or is not smaller than what it replaces is +thrown away and the conversation is left as the free steps left it. While it is being +written the status row shows `tidying` with a clock. diff --git a/internal/manual/chat/hints-and-tips.md b/internal/manual/chat/hints-and-tips.md index 54114b40f3..ef1158da5e 100644 --- a/internal/manual/chat/hints-and-tips.md +++ b/internal/manual/chat/hints-and-tips.md @@ -116,7 +116,7 @@ that is true for you gets its turn before any repeats, and a tip that stops bein stands down at once for the next. (A conversation's keys row does not take turns: it ranks, and the first tip in the list that is true for you there is the one it says.) Nothing outranks anything — with one exception. **A tip that has just become true jumps the queue**: when a conversation crosses half its context -window, `/compact summarizes the conversation now` is said next rather than forty minutes +window, `/compact shortens the conversation now` is said next rather than forty minutes later when the ring comes round. It jumps once and then takes its turn like the rest. ## Every hint codeaf can show, and what makes each one go away @@ -127,7 +127,7 @@ build if the two disagree), so a tip you saw is on it word for word. **Starting work** -- `/compact summarizes the conversation now` — when the conversation passes half its +- `/compact shortens the conversation now` — when the conversation passes half its context window. Retired when a `/compact` finishes. - `/cost says what this conversation has spent` — once the conversation has spent about ten cents. Retired when you run `/cost`. diff --git a/internal/manual/chat/models-and-cost.md b/internal/manual/chat/models-and-cost.md index a695887562..9f5e813230 100644 --- a/internal/manual/chat/models-and-cost.md +++ b/internal/manual/chat/models-and-cost.md @@ -1463,6 +1463,10 @@ answer: it cannot think harder than its own ceiling. remembered, so it costs one rejected request for that model and never a failed reply. - A model whose catalog row says it takes no reasoning knob at all is sent nothing about thinking. +- On a small context window the budget shrinks to the room the conversation leaves beside + an answer. When less than 1,024 tokens of thinking would fit, the budget is dropped and + the rung travels as `high`, so `xhigh` on a 32,000-token window is never refused for the + size of its own thinking. These rungs are not the same notation as a thinking level written onto a pin (`moonshotai/kimi-k3:high`), which still takes only `low`, `medium` and `high`. @@ -1826,7 +1830,7 @@ These are the sentences and what each one means. | --- | --- | --- | | `that model is not being served any more` | the router has no machines behind that model id at all | moves to your next fallback model at once, with no tries wasted | | `your key was not accepted for this model` | a key that is missing, not permitted for this model, or out of balance | stops and tells you — no machine, shape or model changes this | -| `this conversation got too long for the model` | the transcript is past the model's window | shortens the conversation once and asks the same question again | +| `this conversation got too long for the model` | the transcript is past the model's window | reduces the request and retries within a bounded recovery episode | | `this conversation is too long for the model even after shortening it` | it still did not fit | stops; start a new conversation, or `/model` to one with a bigger window | | `the request could not be sent as it was` | the router read the request itself and refused it | the request was already retried with its optional parts taken off; nothing else will help | | `nothing came back from the model — asking again` | a reply arrived with no words and no tool call | asks again on the same budget as any other failure | @@ -2656,42 +2660,42 @@ An unknown window has no threshold at all. ## The most tokens one request can carry — the model's own window, and the ceiling an endpoint puts on it -**The threshold follows the model's own window.** On a model claiming 1,310,720 tokens, -compaction fires at **1,114,112** — not at some smaller figure of codeaf's choosing. On the -default 128,000-token window it fires at 108,800. The line is always -`window − max(15% of window, 16384)`, and `window` is what the model card says. - -There used to be a flat ceiling of 256,000 over every model alike, and it made a -million-token model fold exactly like a small one — nineteen passes in one two-and-a-half -hour run, each at around a hundred thousand tokens, each one throwing the provider's prompt -cache away. That ceiling is gone. - -**What can still lower it is an endpoint refusing.** If a provider answers that a request -would not fit, codeaf writes down how big that request was and never trusts that model past -that size again — in this conversation from the next check onward, and on this machine for -good, because the note is kept in `model-quirks.json` beside your other settings. That is -the one thing allowed to contradict a model card, and it is the only thing: a published -window is a claim, and a refusal is a measurement. - -It exists because a claim can be very wrong. A session on -`~deepseek/deepseek-v4-flash-latest` — a row claiming 1.3M tokens — grew to 386,309 tokens -without compaction firing once, and what came back at that size was the model's own template -turned inside out rather than an answer. That now costs one turn on that model on this -machine, instead of costing every model with real room every turn for ever. - -What the status line reports is still the model's own window, because that line is -describing the model. - -**A request that would not fit is never sent.** Immediately before each request goes out, -a transcript already past the trusted window is compacted first — and unlike the ordinary -pass, this one runs **even when automatic compaction is switched off**. Fitting is not a -preference. Nothing is truncated and nothing of yours is dropped; it is the same pass -`/compact` runs, and every message you typed survives it. - -**Accuracy note.** codeaf also carries a shared context-budget package with a 60%-fill rule, -a 160k working set and a 250% reuse law. **That package is not used by this chat.** Its -consumer is the sub-harness leaf sizing elsewhere in codeaf. The chat's own law is the one -above — do not describe this conversation as filling to 60%. +The model catalog is the starting window. Before sending a conversation request, codeaf +checks the encoded messages, tool schemas and replayed reasoning, the output allowance +including thinking, and a safety margin. It uses the smallest known context window among +endpoints the request can reach. A strict provider pin excludes other endpoints; an +advisory order does not. A request that carries tools is measured against endpoints +known to take tools; after a pinned endpoint refuses it, that endpoint's learned +limit also sizes later tool requests. A limit learned from a refusal counts for +30 minutes even when the endpoint is absent from the sheet. With default routing +the router can still send a tool request to an endpoint that takes no tools. If +that endpoint refuses it as too long, codeaf sends +the same request once more before shortening anything; its window does not size later +tool requests unless that endpoint is pinned. + +An endpoint's explicit total limit is remembered by base URL, model and provider in +`model-quirks.json` for 30 minutes, then learned again if it still holds. A rejected +prompt's length is **not** a total context limit. Old model-wide `served_window` guesses +are no longer used to size requests. + +## How much room is left for the answer — the output cap, max_tokens and the safety margin + +When no output cap was requested, the total output allowance is the smaller of the answer +room setting and a quarter of the effective window. It is sent as the request's +`max_tokens` only when it does work — a thinking budget, a window too small for an ordinary +answer, or an endpoint that stated its window when it refused — and an ordinary request +leaves the output cap to the endpoint, since an unasked cap can rule out an endpoint whose +own cap is lower. It may shrink to fit, while normally +keeping at least 512 output tokens (an eighth of a very small window). A thinking budget +shrinks to fit beside room for an answer, and is dropped below 1,024 tokens, so thinking +alone never makes a request too long. The safety margin is 5% of the window, bounded between +512 and 8,192 tokens. These are estimates, not a provider tokenizer. + +If the request still cannot fit, it is shortened before sending. If protected content +cannot fit either, codeaf stops locally with `context needs shortening before sending`, +naming the input, the short answer it kept room for and the safety margin, and suggests +compaction or a larger-context model. This guard stays on with `--no-compact`. +The status line still shows the catalog window; an endpoint may have a smaller limit. ## A task or a worker on another model gets that model's window @@ -2711,10 +2715,11 @@ smallest window this surface routes to and the safe direction for a guess to be ## What happens before the conversation is compacted -codeaf does not jump straight to summarizing. There are rungs before it. +codeaf uses mechanical reductions first; a summary is written only when they are not +enough (see *What a compaction pass keeps*). -**During one long turn, tool output has its own working-set bound.** Once the live request -estimate crosses **64,000 tokens** — or half the trusted context window when that is smaller +**During one long turn, tool output has its own working-set bound.** Once tool observations from the turn +cross **64,000 tokens** — or half the trusted context window when that is smaller — codeaf replaces already-seen tool results from that turn with the same readable pointer lines described below. It works in whole tool batches, oldest first, while leaving the latest **20,000 tokens** verbatim (capped at a quarter of a smaller window). The result from @@ -2795,15 +2800,7 @@ are appended at the *end* of the conversation and never written into the system because a system message that changed would make every message behind it new again, while a note at the end costs only the note. -**Rung 2 — page images.** Instead of summarizing the part being dropped, it can be -photographed: rendered verbatim to monospaced page images that the model reads back. No model -call, nothing paraphrased. This rung is chosen only when you gave `/compact` no focus, there -is a workspace, there is page budget, and the model in use can read images. Pages are 120 -columns by 64 lines, greyscale, deterministic, and footed -` | context page 1 of 4`. The ceiling is **8 pages**; anything past it is folded to a -marker after the pages. - -**Rung 3 — the fold.** If the transcript is still too big after stubbing, the oldest +**The fold.** If the transcript is still too big after stubbing, the oldest **assistant** work is replaced by one marker line. It is not a summary: nothing is described and nothing is decided. @@ -2816,66 +2813,79 @@ nothing else can reconstruct, so the fold walks past them and takes only the ass ## What a compaction pass keeps -**A compaction asks no model, costs nothing, and takes no time you can feel.** There is no -summarizer behind it — there was one, and it was deleted. It paid a model to write prose -about the text it was about to throw away, at the worst possible moment, and the loss was -unrecoverable because the transcript the prose came from went with it. +Compaction first asks no model: tool results can become pointers to their full bytes, and +older assistant work can become a marker naming the saved journal. Only when that cannot +reach the line the pass needs does the conversation's own model write a summary of the +oldest part, your older messages included, marked `[context compacted]`. The system prompt +and your three most recent messages, with everything after them, stay word for word +when no cut can reach the line. If a cut can reach it, codeaf chooses the one that +keeps the most. Recovery from a refused request can keep fewer to reclaim the +missing room; your latest message always stays word for word. + +Routine cleanup keeps the latest 20,000 tokens, capped at a quarter of the window, and +protects the running turn. `/compact` and necessary request-size recovery can also fold +older completed batches within the running turn. They keep a 4,096-token tail, capped at +an eighth of the window, and always keep the newest assistant/tool batch whole. A batch +that alone exceeds the allowance stays whole; pending tool calls are not discarded. + +Manual compaction does not wait for the automatic threshold. It folds eligible history +outside that smaller tail. Necessary recovery takes enough older work to buy headroom, +then checks the newly assembled request again. Those two free steps paraphrase nothing; +a summary paraphrases the older part, while the originals stay in the session journal. +Removed reasoning leaves together with the assistant message it belongs to. + +`nothing to compact — your messages and recent work are kept` means no eligible material +could be reduced. It does not mean the whole request fits. A genuinely concurrent pass +may return `session: a compaction pass is already running`. -What replaces it is two mechanical passes over messages this session already has: tool -results become pointers to their own bytes, and then the oldest assistant work becomes one -marker line naming where the whole of it can still be read. +## What happens when the conversation gets too long — when compaction happens by itself -What the model is handed instead of a summary is the **state card** — what `track` and -`commit` recorded — which rides in the system prompt on every turn and is kept up to date -after each one. So what the conversation is about is never paraphrased, because it was never -written as prose in the first place. +Routine compaction starts after a step crosses the automatic threshold. Necessary +compaction also runs when the assembled provider request cannot fit, regardless of +`--no-compact`. It considers the input, tool definitions, replayed reasoning, output +allowance and safety margin together. -A pass can decline: `session: nothing to compact` (everything already fits in the tail), or -`session: a compaction pass is already running`. +A provider can still reveal an unknown or changed limit. codeaf recognizes that overflow, +reads a reported total limit when available, and retries only after reducing the request, +learning a changed limit or correcting its input estimate. There are at most **two recovery +attempts per failed generation**. A successful response resets the allowance, so a later +overflow in the same long turn can recover too. Completed tool actions are not rerun. -## What happens when the conversation gets too long — when compaction happens by itself +A pass that reduces history shows `compacting ~84k tokens`, then a `compacted` line +reporting the work folded and the size reduction. Automatic attempts that find nothing +to reduce leave no seam; `/compact` still reports its no-op. Your scrollback and the journal keep the original record. The +model's fold marker names the journal so `read` or `grep` can recover the omitted work. -When the conversation gets too long to fit, nothing is lost and nothing stops: the oldest -part of it is stubbed and folded down to a marker and the recent tail is kept, which is what -compaction is. - -**Nothing is lost is meant literally, and you can go and look.** The session file keeps -every original line, and scrolling up above the boundary is given those rather than the -shortened copy — with one dim line, `· above here the model keeps a shortened record — you -can still read it all`, where the two meet. What shrank is the model's copy, not yours (the -screen page has the whole of that line's meaning, and the limit: a session compacted by an -older codeaf is still drawn from the shortened copy). The fold marker the model sees names -that journal as a real path — `[folded 31 messages · grep or read /home/x/.codeaf/v3/sessions/abc.jsonl, lines 12..40]` -— so codeaf can open the lines that left the window itself. *Where did the folded messages -go* on the compacting page is the whole of that. - -Four ways a pass starts: - -- **Automatically**, after any step where the estimate is over the threshold. A failed pass is - not a failed turn. -- **Just before a request that would not fit**, when the transcript is already past the - 256,000-token ceiling. This one runs **even when automatic compaction is switched off** — - the switch governs headroom, and fitting is not headroom. -- **On a context-overflow error from the provider**, once per turn. This one runs **even when - automatic compaction is switched off** — the switch governs the automatic pass, not the - recovery from a request the provider has already refused. Overflow errors are never retried. -- **On demand**, when you ask for it. - -While a pass runs you see `compacting ~84k tokens` (`~842` under a thousand). On success: -`compacted from ~84k tokens, kept last ~20k`, or with page images -`compacted from ~84k tokens to 4 page images, kept last ~20k`. On failure: -`compaction failed · context unchanged`. A failed pass always settles its row. +Recovery from a refused request can also end with a summary of the oldest part of the +conversation, written by the conversation's own model. If your most recent message and the +newest working batch still cannot fit, shortening stops and the request is refused locally. A new conversation or a larger-context model is needed. +An unfamiliar endpoint can still reject a first request: estimates and published limits +cannot guarantee that every provider's initial response succeeds. ## /compact — compacting now -`/compact` compacts the conversation on demand. It notes `compacting…` immediately and runs -the pass off the loop, so the surface stays alive. - -**Success is silent.** There is no "done" message — a compaction that worked simply leaves the -conversation shorter. A failure comes back as `compact failed: ` followed by the error. - -**It costs nothing and asks no model**, so there is no reason not to run it, and no `compaction` -role in settings to point at a model for it. +`/compact` runs a reduction immediately, even below the automatic threshold. It notes +`compacting…` and works off the input loop. Older completed work becomes pointers to the +full journal first, which makes no model call. If enough older conversation remains, +the conversation's own model summarizes it, your older messages included — in the same +pass, so one `/compact` goes as far as it can. + +Your three most recent messages and everything after them stay word for word when +no cut can reach the line. If a cut can reach it, codeaf chooses the one that keeps +the most. Recovery from a refused request can keep fewer to reclaim the missing +room; your latest message and system prompt always stay. +Success reports `⚭ compacted · about N to M tokens` when the measured count fell, or +`⚭ compacted` without a size when it did not; those figures are estimates. A no-op says +`nothing to compact — ` and why. A pass still running after five minutes says +`still compacting — it is taking longer than usual and finishes on its own; the token count in the status line drops when it lands`. Other +failures say `compact failed: ` followed by the reason. + +There is no separate compaction model or summarization setting. The smaller retained tail and the +explicit reduction distinguish this command from routine automatic cleanup. + +`/compact` ends with a summary written by the conversation's own model whenever at least +about 1,000 tokens of older conversation are left to summarize; its cost counts toward the +session's spending like any other call. ## Turning automatic compaction off @@ -2885,7 +2895,7 @@ text reads `never compact automatically`. What it turns off is exactly the automatic threshold check. Still working: - `/compact`, when you ask for it, and -- the recovery pass when the provider itself refuses a request as too long. +- the request-size guard and recovery when the provider refuses a request as too long. With a `--host` remote launch the flag is **refused rather than ignored**, because it cannot travel to the other machine. @@ -4032,9 +4042,19 @@ words: reports. - **`full`** sends everything whatever the model reports. -**A change lands the next time codeaf starts.** The profile is settled once when -a conversation opens, because it decides the page and the tool list every -request in that conversation is sent with. +**Changes to this setting land the next time codeaf starts.** The launch preference +stays fixed. With `auto`, selecting a model below 32,000 tokens changes the prompt +and default tool list together before the next request; selecting a larger model +restores the full profile. An answer already in flight keeps its original shape. +Explicit `lean` and `full` choices override that automatic switching. + +Capabilities you explicitly loaded and connected service tools remain available +across a switch. Consequently, a conversation with many loaded tools can still be +too large for the smaller model. A profile change does not summarize history. + +The lean profile is still too large for some 4k–8k models. Automatic selection is +not a promise that every model can fit the prefix or use tools; the final request +guard still checks the assembled request. **`CODEAF_PROMPT_PROFILE` still pins it for one launch, over the row.** Put `CODEAF_PROMPT_PROFILE=lean` or `CODEAF_PROMPT_PROFILE=full` in front of the @@ -4059,3 +4079,42 @@ check, and completed spending resets at the local date boundary. Calls already in flight may finish; this is not an atomic reservation across processes. Use `/budget` to change the limit. Per-conversation and per-task limits still apply separately. + +## A model that can't use tools — what retry "removed tools" meant, "can't use tools, so it answers without them", "sent without tools" + +Some models take no tool calls at all. When the model catalog says so (the model's +published parameters do not include tools), codeaf leaves every tool off the request from +the start and says so once per model, the first time: + +``` +microsoft/phi-4 can't use tools, so it answers without them — it cannot read, search or change files +``` + +The conversation still works as a chat: the model answers from what it knows and from what +you paste in. It cannot open files, run commands or look anything up, and it will say so +if you ask it to. It is also given a **short chat page** as its instructions instead of +codeaf's working page — about 2,000 characters rather than 20,000 or more, because the +working page is almost all about tools — and memory is off, as on a small window. Your +project's instruction files (AGENTS.md, CLAUDE.md) are not sent to it. The request-size +check counts only what is really sent. Pick a model that takes tools (`/model`) when you +need it to work in your files: the next message goes out with the working page and every +tool back, including any group you had loaded. A lean or full setting you chose still +applies to every model that can use tools. + +A model the catalog does not know is still sent its tools and its working page. If no +provider serving it accepts them, the retry line says so — it used to read +`Retry 1/1: removed tools`, which did not say why. If that retry is answered, codeaf sends +the model no tools for the next 30 minutes, so the refusal is not paid on every message, +and the one-time line above follows; a retry that also failed teaches nothing: + +``` +Retry 1/1: sent without tools, which no provider serving this model accepts — it cannot read, search or change files on this answer +``` + +## Switching back to a model without tools after a file read — earlier calls and results + +When you switch to a model that cannot use tools after another model used one, +the earlier call and its result are carried as readable conversation text. The +model can discuss that result, but it cannot make a new tool call itself. +Switching back to a tool-capable model restores the original tool-call history +on its request. diff --git a/internal/manual/chat/running-on-another-machine.md b/internal/manual/chat/running-on-another-machine.md index f7db57a017..a3d803deca 100644 --- a/internal/manual/chat/running-on-another-machine.md +++ b/internal/manual/chat/running-on-another-machine.md @@ -856,7 +856,8 @@ While that is happening the status line says, quietly: reconnecting to devbox — trying for up to 5 minutes ``` -Every call still gives up after 10 seconds, so a dead pipe never leaves your terminal frozen. +Ordinary calls give up after 10 seconds. A remote `/compact` waits up to five minutes +for a summary to finish; the chat stays responsive while it waits. If it cannot get back at all, you see the sentence you always saw: diff --git a/internal/manual/chat/screen.md b/internal/manual/chat/screen.md index 7cdcf04ca1..42065edcf5 100644 --- a/internal/manual/chat/screen.md +++ b/internal/manual/chat/screen.md @@ -672,8 +672,9 @@ Both halves of the line are true and neither one covers for the other: the words it was said in — your messages, the replies, the tool calls and their whole output. Nothing was thrown away. - **The model does not.** Above that line the model is working from a shortened version: - old tool results became one-line pointers to the files that hold them, and long runs of - its own earlier work became a single line saying how much went. So if you ask about + old tool results became one-line pointers to the files that hold them, long runs of + its own earlier work became a single line saying how much went, and — when that was not + enough — the oldest part of the conversation became a summary it wrote. So if you ask about something above the line, it may answer from something shorter than what you are looking at — ask it to `read` the file, or paste the part you mean back in. @@ -3465,7 +3466,7 @@ conversation you are in, the room you are standing in — and the **copy span's* are facts about the session rather than about a pointer, and they are true whoever is reading. What is dropped is the quieter background the pointer and the cursor share. -## Two other things that move on screen +## The pulsing ellipsis and "still working" — what moves on screen while a turn waits **The pulsing ellipsis, and the `still working` fallback.** When a turn is running and nothing else on screen is moving, two spaces then a pulsing ellipsis cycles `·` → `··` → @@ -3489,12 +3490,21 @@ left when neither of the lines above knows anything. It says "still working" and "trying again" — a silence is only a silence to this suffix, and the words change to `trying again` solely when the request really was cut and re-sent, which is said outright. -**The compaction mark.** A compaction is drawn while it runs and left as a rule once it -lands, so the conversation never silently loses its middle. Running, it reads +## The compaction line — what "⚭ compacted" means, and why a summary's line stays after the turn + +A compaction is drawn while it runs and left as one quiet line +once it lands, so the conversation never silently loses its middle. Running, it reads `⠙ compacting ~84k tokens · 6s` — a braille spinner on the same grid as the tool -spinners, dim, with a count-up. Settled, it becomes a centred rule: -`───── ⚭ compacted from ~84k tokens · took 6s ─────`. The duration is dropped under one -second. It is never painted the question hue, because nobody is being asked anything. +spinners, dim, with a count-up. Settled, it becomes a dim line in the notes' lane: +`· ⚭ compacted · summarized 4 messages · ~31k → ~13k tokens · full record in the session journal · took 6s`. The duration is +dropped under one second. It is never painted the question hue, because nobody is being +asked anything. **A pass that wrote a summary stays after the turn ends**: when the turn's +work folds behind `▸ worked`, its line stands outside the fold — above the answer when it +ran mid-turn, under it when it ran at the end — because it rewrote your own words in the +model's copy of the conversation. A pass that only folded or stubbed is ordinary machinery +and folds with the rest of the turn; `ctrl+e` opens it. A `/compact` +you run yourself leaves the same mark: `⚭ compacted · about N to M tokens` when the +measured count fell, or `⚭ compacted` without a size when it did not. ## What is it doing right now — connecting, first word, thinking, writing, paced, trying again diff --git a/internal/manual/chat/when-the-connection-drops.md b/internal/manual/chat/when-the-connection-drops.md index fac1dcb933..6fb9fcbab9 100644 --- a/internal/manual/chat/when-the-connection-drops.md +++ b/internal/manual/chat/when-the-connection-drops.md @@ -294,8 +294,8 @@ sitting at, and a hundred of those against a machine that is switched off is you working for nothing. A blip is caught by the first retry; a machine that is really gone is not worth hammering. -Every call codeaf makes over a connection already gives up after 10 seconds, so nothing -about a dead link can leave your terminal frozen while this is going on. +Ordinary calls over a connection give up after 10 seconds. A remote `/compact` waits up +to five minutes for a summary to finish; the chat stays responsive while it waits. ## It said something "fell over once and will be tried again" diff --git a/internal/manual/chat_test.go b/internal/manual/chat_test.go index ca4546e9e9..bd9152fc5e 100644 --- a/internal/manual/chat_test.go +++ b/internal/manual/chat_test.go @@ -1010,6 +1010,13 @@ func TestTheChatManualAnswersTheQuestionsPeopleAsk(t *testing.T) { {"where did the folded messages go", "compacting-over-and-over"}, {"how do I get the compacted text back", "compacting-over-and-over"}, {"what happened to the earlier messages", "compacting-over-and-over"}, + // Summaries came back as compaction's last rung on 2026-09-28. + {"does codeaf summarize my conversation", "compacting-over-and-over"}, + {"when does compaction write a summary", "compacting-over-and-over"}, + // A tool-less model is told once rather than retried every turn (2026-09-28). + {"why does it say the model can't use tools", "models-and-cost"}, + {"what does retry removed tools mean", "models-and-cost"}, + {"what does the summary keep", "compacting-over-and-over"}, {"does it work on a narrow phone width terminal", "screen"}, {"why is my table cut off", "screen"}, {"why does the receipt say the compiler supplied no reading", "adaptive-runs"}, diff --git a/internal/namelaw/namelaw_test.go b/internal/namelaw/namelaw_test.go index 733c6ec4d6..f04bb1ce07 100644 --- a/internal/namelaw/namelaw_test.go +++ b/internal/namelaw/namelaw_test.go @@ -226,7 +226,7 @@ func TestW7ThePromptNamesCodeafOnceAndNamesNoRetiredProduct(t *testing.T) { if err != nil { t.Fatal(err) } - wantFiles := []string{"bashrules.md", "bashtask.md", "bashworker.md", "discipline.md", "divide.md", "fanout.md", "landing-answer.md", "program-outcome.md", "quick.md", "revise.md", "runask.md", "runsummary.md", "shape.md", "system.md", "worker.md"} + wantFiles := []string{"bashrules.md", "bashtask.md", "bashworker.md", "chat.md", "discipline.md", "divide.md", "fanout.md", "landing-answer.md", "program-outcome.md", "quick.md", "revise.md", "runask.md", "runsummary.md", "shape.md", "system.md", "worker.md"} var gotFiles []string var corpus []byte for _, entry := range entries { diff --git a/internal/provider/attribution_wire_test.go b/internal/provider/attribution_wire_test.go index ffc2512caf..d2ff3f098f 100644 --- a/internal/provider/attribution_wire_test.go +++ b/internal/provider/attribution_wire_test.go @@ -264,7 +264,7 @@ func TestOwnClientKeepsTheRefusalLadder(t *testing.T) { // The ladder narrated itself and ended on the rung that takes tools off. mu.Lock() defer mu.Unlock() - if len(notices) == 0 || !strings.Contains(notices[len(notices)-1], "removed tools") { + if len(notices) == 0 || !strings.Contains(notices[len(notices)-1], "sent without tools") { t.Fatalf("retry notices = %#v", notices) } // More than one request was made, and every one of them was ours. diff --git a/internal/provider/client.go b/internal/provider/client.go index eee6365684..77f860c0a8 100644 --- a/internal/provider/client.go +++ b/internal/provider/client.go @@ -177,6 +177,9 @@ type Client struct { // carry (withdrawn.go). It is beside `encodes` for the same reason: a fact // about one router and one account, over the span of one conversation. withdrawn withdrawnMemo + // toolless is which models this client has said it sends no tools to + // (toolless.go). + toolless toollessMemo // encodes is what this client already knows its transcript and its tool // block serialize to (memo.go). It changes nothing about the bytes and is // carried per client because a transcript belongs to a conversation. @@ -416,7 +419,8 @@ type callKnobs struct { relaxed relaxSet // reasoning is aligned with the request's messages. It stays outside the SDK // values because ai.Message has no reasoning fields of its own. - reasoning []MessageReasoning + reasoning []MessageReasoning + contextBudget ContextBudget // noProvider takes the `provider` object OFF this one encode entirely, and // it is set by exactly one caller: the single widened retry that asks // whether a base's 400 was about the field at all (endpoints.go's @@ -490,17 +494,18 @@ func (k callKnobs) carriesTheDemand() bool { func knobsFrom(ctx context.Context) callKnobs { knobs := callKnobs{ - cacheKey: CacheKeyFrom(ctx), - effort: effortFrom(ctx), - role: RoleFrom(ctx), - intent: routingIntentFrom(ctx), - lambda: valueOfTimeFrom(ctx), - horizon: callHorizonFrom(ctx), - hedgeLane: hedgeLaneFrom(ctx), - reasoning: MessageReasoningFrom(ctx), - refused: &refusedHere{}, - retryAvoid: RetryAvoidFrom(ctx), - trace: newCallTrace(), + cacheKey: CacheKeyFrom(ctx), + effort: effortFrom(ctx), + role: RoleFrom(ctx), + intent: routingIntentFrom(ctx), + lambda: valueOfTimeFrom(ctx), + horizon: callHorizonFrom(ctx), + hedgeLane: hedgeLaneFrom(ctx), + reasoning: MessageReasoningFrom(ctx), + contextBudget: contextBudgetFrom(ctx), + refused: &refusedHere{}, + retryAvoid: RetryAvoidFrom(ctx), + trace: newCallTrace(), } // The choice this call was already made on, if it was. See // [Client.withLaneChoice]: it is carried rather than recomputed because it @@ -595,6 +600,9 @@ func (c *Client) sendShaped(ctx context.Context, request *ai.Request, knobs call return nil, withdrawnRefusal(model) } noteModelTried(ctx, model) + // A MODEL THE CATALOG SAYS TAKES NO TOOLS IS SENT NONE (toolless.go), before + // any body is encoded, so the size check measures what really goes out. + knobs = c.leaveOffTools(ctx, model, knobs, len(request.Tools) > 0) response, err := c.sendRecovered(ctx, request, knobs, stream) // AN ANSWER MEANS IT IS CARRIED AGAIN. A memo nothing clears takes a model // away for the life of the process on the strength of one bad minute. @@ -629,6 +637,41 @@ func (c *Client) sendRecovered(ctx context.Context, request *ai.Request, knobs c // recoverFromPacing). The layer that owns the turn owns the model. return nil, err } + // Simple routing has no hard parameter filter. The router can send a tool + // request to a tool-less endpoint with a smaller window than local sizing + // used. Re-send once before the conversation pays for a summary; a second + // overflow goes to its usual recovery owner without another loop. ONLY A + // ROUTER CAN ANSWER A RESEND FROM ANOTHER ENDPOINT: a direct base is the one + // endpoint, and resending it the same request would pay a refusal twice. + if response != nil && endpointRefusalStatus(response.StatusCode) && knobs.contextBudget.Window > 0 && !c.config.Direct && c.baseServesLanes() { + peek, readErr := io.ReadAll(io.LimitReader(response.Body, maxErrorPeek)) + if readErr == nil { + prefs := refusedWirePreferences(response) + choice, chosen := laneChoiceFromContext(ctx) + if failure, ok := RefusalFrom(apiError(response.StatusCode, peek)); ok && failure.Overflow && failure.FromUpstream() && + failure.ContextLimit > 0 && failure.ContextLimit < c.servingWindow(c.modelFor(request), prefs, knobs.contextBudget.Window, len(request.Tools) > 0) { + if prefs != nil && len(prefs.Only) == 1 || chosen && choice.Pinned && len(choice.Only) == 1 { + c.rememberContextLimit(c.modelFor(request), failure) + response.Body = rewound(peek, response.Body) + return response, nil + } + c.record(recordFacts{ctx: ctx, request: request, knobs: knobs, stream: stream, + attempt: c.attemptsSoFar(knobs), began: began, status: response.StatusCode, + err: apiError(response.StatusCode, peek), responseBody: peek}) + response.Body.Close() + retried, retryErr := c.sendRepaired(ctx, request, knobs, stream) + c.rememberContextLimit(c.modelFor(request), failure) + if retryErr != nil { + return nil, retryErr + } + response = retried + } else { + response.Body = rewound(peek, response.Body) + } + } else { + response.Body = rewound(peek, response.Body) + } + } if !endpointRefusalStatus(response.StatusCode) { return response, nil } @@ -766,7 +809,7 @@ func (c *Client) sendRecovered(ctx context.Context, request *ai.Request, knobs c func (c *Client) sendRepaired(ctx context.Context, request *ai.Request, knobs callKnobs, stream bool) (*http.Response, error) { body, err := c.encodeRequest(request, knobs) if err != nil { - return nil, fmt.Errorf("marshal request: %w", err) + return nil, encodeFailure(err) } began := logNow() response, err := c.send(ctx, request, knobs, body, stream) @@ -842,7 +885,7 @@ func (c *Client) resend(ctx context.Context, request *ai.Request, knobs callKnob refused.Body.Close() body, err := c.encodeRequest(request, knobs) if err != nil { - return nil, fmt.Errorf("marshal request: %w", err) + return nil, encodeFailure(err) } return c.send(ctx, request, knobs, body, stream) } @@ -954,7 +997,17 @@ func (c *Client) newRequest(messages []ai.Message, options []ai.Option) (*ai.Req // be served this way says so on the wire — a gateway that takes `stream: true` // and answers one whole JSON completion — and [Client.unstreamable] remembers it // from what actually happened, so the fallback is a memo rather than a guess. -func (c *Client) CompleteWithMessages(ctx context.Context, messages []ai.Message, options ...ai.Option) (*ai.Response, error) { +func (c *Client) CompleteWithMessages(ctx context.Context, messages []ai.Message, options ...ai.Option) (response *ai.Response, err error) { + // Every failed completion can teach the next encode, including the retry + // after an empty thinking-only answer. + defer func() { + if failure, ok := RefusalFrom(err); ok { + request, requestErr := c.newRequest(messages, options) + if requestErr == nil { + c.rememberContextLimit(c.modelFor(request), failure) + } + } + }() ctx = WithPlanOverflowGuard(ctx) observer := streamObserverFrom(ctx) response, relearned, err := c.completeWithMessagesStreaming(ctx, observer, messages, options...) @@ -2544,6 +2597,12 @@ type APIError struct { // envelope's own `code`, with the sentence kept only as a hint for a body // that carries neither ([overflowRefusal]). Overflow bool + // Context facts are optional evidence, parsed once at the refusal boundary. + ContextLimit int + InputTokens int + OutputTokens int + Local bool + BudgetChanged bool // Code is the error envelope's `code`, as text. The router types that field // as a number, as a string, and sometimes omits it, so it is normalised here // once rather than decoded at each reader. @@ -2776,6 +2835,7 @@ func apiError(status int, payload []byte) error { // built and is therefore true of every APIError this build makes, including // the ones a stream raises in-band. failure.Overflow = overflowRefusal(status, failure.Code, failure.Message, failure.Raw) + readContextLimit(failure) return failure } diff --git a/internal/provider/contextbudget.go b/internal/provider/contextbudget.go new file mode 100644 index 0000000000..ed2e4bf335 --- /dev/null +++ b/internal/provider/contextbudget.go @@ -0,0 +1,277 @@ +package provider + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "regexp" + "strconv" + "strings" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + lanes "github.com/Agent-Field/codeaf/internal/lane" +) + +// ContextBudget describes the conversation whose next request is being encoded. +// The provider applies it after tool schemas, replayed reasoning and routing +// preferences have been assembled, so the check measures what will be sent. +type ContextBudget struct { + Window int + Reserve int + PromptFloor int +} +type contextBudgetKey struct{} + +func WithContextBudget(ctx context.Context, budget ContextBudget) context.Context { + return context.WithValue(ctx, contextBudgetKey{}, budget) +} +func contextBudgetFrom(ctx context.Context) ContextBudget { + budget, _ := ctx.Value(contextBudgetKey{}).(ContextBudget) + return budget +} + +// ContextLimit is an endpoint's stated total window, never a rejected prompt's +// size. The key includes the account's base URL and the serving endpoint. +type ContextLimit struct { + Base string `json:"base"` + Model string `json:"model"` + Provider string `json:"provider,omitempty"` + Tokens int `json:"tokens"` + At time.Time `json:"at,omitempty"` +} + +// A router's endpoint list changes during a long session. Half an hour is the +// same hold used for a serving refusal in the lane sheet: long enough to avoid +// relearning on each turn, short enough to let a changed endpoint recover. +const servingFactHold = 30 * time.Minute + +func contextLimitKey(base, model, endpoint string) string { + return strings.TrimRight(strings.ToLower(base), "/") + "\n" + normalizeModel(model) + "\n" + strings.ToLower(strings.TrimSpace(endpoint)) +} +func (c *Client) rememberContextLimit(model string, failure *APIError) { + if failure == nil || !failure.Overflow || failure.Local || failure.ContextLimit <= 0 { + return + } + limit := ContextLimit{Base: c.config.BaseURL, Model: normalizeModel(model), Provider: failure.Provider, Tokens: failure.ContextLimit} + changed := storeContextLimit(limit) + failure.BudgetChanged = changed + // The same limit is still fresh evidence; its new date must survive restart + // without treating it as a new budget for recovery. + quirks.persist() +} + +// storeContextLimit records one endpoint's stated window and reports whether +// it is news. The memo's lock is released by a defer, so a panic inside cannot +// leave every later request waiting on it. +func storeContextLimit(limit ContextLimit) bool { + now := time.Now() + if limit.At.IsZero() || limit.At.After(now) { + limit.At = now + } + key := contextLimitKey(limit.Base, limit.Model, limit.Provider) + quirks.mutex.Lock() + defer quirks.mutex.Unlock() + if quirks.contextLimits == nil { + quirks.contextLimits = make(map[string]ContextLimit) + } + old, known := quirks.contextLimits[key] + quirks.contextLimits[key] = limit + return !known || old.Tokens != limit.Tokens || !old.At.After(time.Now().Add(-servingFactHold)) +} + +// servingWindow takes the smallest known window among endpoints this request +// can reach. A strict pin excludes other endpoints; an advisory order does not. +// This is a read of the existing sheet and memo, with no network work. +// +// A TOOL REQUEST USES ONLY TOOL ENDPOINTS FOR LOCAL SIZING. Ranked routing +// sends `require_parameters`, but default Simple routing sends no provider +// object and the router can still send tools to a tool-less endpoint. A relayed +// overflow from that endpoint gets one resend before the caller compacts; its +// smaller learned limit does not cap later tool requests unless the person +// pinned that endpoint. +func (c *Client) servingWindow(model string, prefs *providerPrefs, claimed int, carriesTools bool) int { + window, _ := c.servingWindowStated(model, prefs, claimed, carriesTools) + return window +} + +// servingWindowStated is [Client.servingWindow] plus whether the window in +// force is one an endpoint STATED in a size refusal. That refusal is evidence +// the endpoint's own default answer did not fit behind the prompt, which is +// what makes an explicit ceiling worth sending ([Client.budgetWire]). +func (c *Client) servingWindowStated(model string, prefs *providerPrefs, claimed int, carriesTools bool) (int, bool) { + window, stated := claimed, false + take := func(tokens int, fromRefusal bool) { + if tokens > 0 && (window <= 0 || tokens < window) { + window, stated = tokens, fromRefusal + } else if fromRefusal && tokens > 0 && tokens == window { + stated = true + } + } + accepts := func(endpoint string) bool { + if prefs == nil || endpoint == "" { + return true + } + return !namesEndpoint(prefs.Ignore, endpoint) && (len(prefs.Only) == 0 || namesEndpoint(prefs.Only, endpoint)) + } + toolSupport := map[string]bool{} + if !c.config.Direct && c.baseServesLanes() { + for _, row := range lanes.Default().Sheet().Rows(laneModel(model)) { + toolSupport[strings.ToLower(strings.TrimSpace(row.ID.Lane))] = row.Facts.Tools + if accepts(row.ID.Lane) && (!carriesTools || row.Facts.Tools) { + take(row.Facts.Context, false) + } + } + } + for _, limit := range storedContextLimits(c.config.BaseURL, model) { + if takesTools, known := toolSupport[strings.ToLower(strings.TrimSpace(limit.Provider))]; accepts(limit.Provider) && (!carriesTools || !known || takesTools || prefs != nil && len(prefs.Only) == 1 && namesEndpoint(prefs.Only, limit.Provider)) { + take(limit.Tokens, true) + } + } + return window, stated +} + +// storedContextLimits is every window the memo holds for one account and +// model, whichever endpoint stated it. +func storedContextLimits(base, model string) []ContextLimit { + want := contextLimitKey(base, model, "") + quirks.mutex.Lock() + defer quirks.mutex.Unlock() + var limits []ContextLimit + for _, limit := range quirks.contextLimits { + if contextLimitKey(limit.Base, limit.Model, "") == want && !limit.At.IsZero() && limit.At.After(time.Now().Add(-servingFactHold)) { + limits = append(limits, limit) + } + } + return limits +} + +// minimumContextAnswer is a useful short answer, not the desired reply size. +// A reserve is a ceiling: requiring thousands of unused output tokens made +// small-window conversations fail even when a substantial answer still fit. +const minimumContextAnswer = 512 + +// minimumContextThinking is the smallest thinking budget worth sending. Below +// it the budget is dropped and the effort travels as its word, which lets the +// endpoint size the pass inside the output ceiling instead; 1,024 is also the +// smallest budget Anthropic's endpoints accept. +const minimumContextThinking = 1024 + +// ContextSafetyTokens leaves room for tokenizer and chat-template differences. +// It grows with small windows and is bounded on million-token models. +func ContextSafetyTokens(window int) int { return min(8192, max(512, window/20)) } + +// budgetWire checks the encoded input and sizes a TOTAL output allowance, +// including thinking. It sends a ceiling when the natural answer would not +// fit behind the prompt, or when a thinking budget needs room above it. It returns the +// thinking budget that fits beside the answer, which is what the request must +// then carry. +// +// THE THINKING BUDGET BENDS TO THE WINDOW; IT NEVER REFUSES A REQUEST. The +// xhigh rung asks for 32,000 tokens of thinking, sized for a 200k window, and +// reserving all of it made a first message on a 32k window "too long" with +// nothing in the conversation to compact (2026-09-28). So the budget shrinks +// to the room the prompt leaves, and a request is refused only when the prompt +// does not leave room for a short answer — the one case compaction can fix. +func (c *Client) budgetWire(request *ai.Request, knobs callKnobs, messages, tools []json.RawMessage, prefs *providerPrefs, ceiling int, hasCeiling bool, thinking int) (int, bool, int, error) { + if knobs.contextBudget.Window <= 0 { + return ceiling, hasCeiling, thinking, nil + } + model := c.modelFor(request) + window, stated := c.servingWindowStated(model, prefs, knobs.contextBudget.Window, len(tools) > 0) + weight := 0 + for _, message := range messages { + weight += len(message) + } + for _, tool := range tools { + weight += len(tool) + } + // Base64 bytes are transport, not input tokens. The image allowance matches + // the conversation estimator; the safety reserve covers template overhead. + for _, message := range request.Messages { + for _, part := range message.Content { + if part.ImageURL != nil { + weight += 4000 - len(part.ImageURL.URL) + } + } + } + prompt := max((weight+3)/4, knobs.contextBudget.PromptFloor) + if !hasCeiling { + ceiling = min(knobs.contextBudget.Reserve, window/4) + if ceiling <= 0 { + ceiling = window / 4 + } + } + room := window - ContextSafetyTokens(window) - prompt + // A useful answer still needs room after thinking. An explicit tiny answer + // is allowed, but an ordinary turn cannot be squeezed to a single token. + floor := min(ceiling, min(minimumContextAnswer, max(1, window/8))) + if thinking > 0 { + answer := min(1024, max(1, window/16)) + if thinking > room-answer { + thinking = room - answer + } + if thinking < minimumContextThinking { + thinking = 0 + } else { + floor = max(floor, thinking+answer) + } + } + if room < floor { + return 0, false, 0, &APIError{Status: http.StatusBadRequest, Code: overflowCode, Overflow: true, Local: true, + ContextLimit: window, InputTokens: prompt, OutputTokens: floor, + Message: fmt.Sprintf("context needs shortening before sending: about %d input tokens plus a %d-token answer and %d safety tokens exceed the %d-token window; compact the conversation or choose a larger-context model", prompt, floor, ContextSafetyTokens(window), window)} + } + // A NORMAL SHORT REQUEST LEAVES THE ENDPOINT ITS OWN COMPLETION DEFAULT, + // as it did before the budget existed: an unasked ceiling is a routing + // filter at the router and can exclude an endpoint whose completion cap is + // lower. The ceiling is sent when it does work — a caller's own, a thinking + // budget that needs room above it, a window too small for the ordinary + // answer, or a window an endpoint stated when it refused a request whose + // default answer did not fit behind the prompt. + return min(max(ceiling, floor), room), hasCeiling || thinking > 0 || stated || room < ceiling, thinking, nil +} + +var contextLimitPatterns = []*regexp.Regexp{ + regexp.MustCompile(`(?i)maximum input length\s*([0-9][0-9,]*)`), + regexp.MustCompile(`(?i)maximum context (?:length|window)(?: is| of|:)\s*([0-9][0-9,]*)`), + regexp.MustCompile(`(?i)(?:context length|context window|context limit)\s*[:=]\s*([0-9][0-9,]*)`), +} +var inputLimitPattern = regexp.MustCompile(`(?i)(?:prompt contains (?:at least )?|input_tokens[^0-9]{1,12}|requested input length\s*)([0-9][0-9,]*)`) +var outputLimitPattern = regexp.MustCompile(`(?i)(?:requested |max_tokens[^0-9]{1,12})([0-9][0-9,]*)(?: output tokens)?`) +var anthropicLimitPattern = regexp.MustCompile(`(?i)prompt is too long:\s*([0-9][0-9,]*) tokens\s*>\s*([0-9][0-9,]*)`) + +func tokenNumber(text string) int { + number, err := strconv.Atoi(strings.ReplaceAll(text, ",", "")) + if err != nil || number <= 0 { + return 0 + } + return number +} + +// readContextLimit extracts optional evidence AFTER overflow classification. +// Missing or changed prose never prevents recovery, and no guessed input count +// is promoted to a durable context-window fact. +func readContextLimit(failure *APIError) { + if failure == nil || !failure.Overflow { + return + } + said := failure.Message + "\n" + failure.Raw + for _, pattern := range contextLimitPatterns { + if match := pattern.FindStringSubmatch(said); len(match) > 1 { + failure.ContextLimit = tokenNumber(match[1]) + break + } + } + if match := inputLimitPattern.FindStringSubmatch(said); len(match) > 1 { + failure.InputTokens = tokenNumber(match[1]) + } + if match := outputLimitPattern.FindStringSubmatch(said); len(match) > 1 { + failure.OutputTokens = tokenNumber(match[1]) + } + if match := anthropicLimitPattern.FindStringSubmatch(said); len(match) > 2 { + failure.InputTokens = tokenNumber(match[1]) + failure.ContextLimit = tokenNumber(match[2]) + } +} diff --git a/internal/provider/contextbudget_ceiling_contract_test.go b/internal/provider/contextbudget_ceiling_contract_test.go new file mode 100644 index 0000000000..12ceefdd7f --- /dev/null +++ b/internal/provider/contextbudget_ceiling_contract_test.go @@ -0,0 +1,31 @@ +package provider + +import ( + "context" + "testing" +) + +func TestContextBudgetOmitsUnneededCeilingButBoundsSmallWindow(t *testing.T) { + client, recorded := newTestClient(t, Config{Model: "review/output", Direct: true}) + for _, test := range []struct { + name string + budget ContextBudget + want bool + }{ + {"large", ContextBudget{Window: 131072, Reserve: 32768}, false}, + {"small", ContextBudget{Window: 16385, Reserve: 4096, PromptFloor: 13903}, true}, + } { + t.Run(test.name, func(t *testing.T) { + ctx := WithContextBudget(context.Background(), test.budget) + _, err := client.CompleteWithMessages(ctx, userMessages("hello")) + if err != nil { + t.Fatal(err) + } + wire := recorded.body(len(recorded.bodies) - 1) + _, has := wire["max_tokens"] + if has != test.want { + t.Fatalf("max_tokens present=%v, want %v: %#v", has, test.want, wire) + } + }) + } +} diff --git a/internal/provider/contextbudget_r3_contract_test.go b/internal/provider/contextbudget_r3_contract_test.go new file mode 100644 index 0000000000..f558c629a5 --- /dev/null +++ b/internal/provider/contextbudget_r3_contract_test.go @@ -0,0 +1,205 @@ +package provider + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + lanes "github.com/Agent-Field/codeaf/internal/lane" +) + +func TestResentOverflowLogsRefusalAndAnswer(t *testing.T) { + read := loggingTo(t) + forgetLanes(t) + body, err := os.ReadFile("testdata/mara-overflow.json") + if err != nil { + t.Fatal(err) + } + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + if strings.HasSuffix(r.URL.Path, "/endpoints") { + _, _ = w.Write([]byte(`{"data":{"endpoints":[{"provider_name":"Mara","context_length":32768,"supported_parameters":[]},{"provider_name":"Wide","context_length":131072,"supported_parameters":["tools"]}]}}`)) + return + } + if calls.Add(1) == 1 { + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write(body) + return + } + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"answered"},"finish_reason":"stop"}]}`)) + })) + defer server.Close() + model := "review/logged-overflow" + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 131072, Reserve: 8192, PromptFloor: 40000}) + if _, err := client.CompleteWithMessages(ctx, userMessages("read"), ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}})); err != nil { + t.Fatal(err) + } + if calls.Load() != 2 { + t.Fatalf("HTTP calls = %d, want 2", calls.Load()) + } + var refused, answered bool + for _, row := range ended(read()) { + refused = refused || row.Status == http.StatusBadRequest + answered = answered || row.Status == http.StatusOK + } + if !refused || !answered { + t.Fatalf("call log has refusal=%v answer=%v", refused, answered) + } +} + +func TestStrictPinLearnsSmallWindowWithoutResending(t *testing.T) { + forgetLanes(t) + body, err := os.ReadFile("testdata/mara-overflow.json") + if err != nil { + t.Fatal(err) + } + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + if strings.HasSuffix(r.URL.Path, "/endpoints") { + _, _ = w.Write([]byte(`{"data":{"endpoints":[{"provider_name":"Mara","context_length":32768,"supported_parameters":[]},{"provider_name":"Wide","context_length":131072,"supported_parameters":["tools"]}]}}`)) + return + } + calls.Add(1) + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write(body) + })) + defer server.Close() + model := "review/pinned-overflow" + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + ctx := WithLaneChoice(context.Background(), lanes.Choice{Only: []string{"Mara"}, Pinned: true}) + ctx = WithContextBudget(ctx, ContextBudget{Window: 131072, Reserve: 8192, PromptFloor: 40000}) + tool := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + _, _ = client.CompleteWithMessages(ctx, userMessages("read"), tool) + if calls.Load() != 1 { + t.Fatalf("strict pin paid %d identical refusals, want 1", calls.Load()) + } + if got := client.servingWindow(model, &providerPrefs{Only: []string{"Mara"}}, 131072, true); got != 32768 { + t.Fatalf("next pinned tool request sized against %d, want 32768", got) + } + _, _ = client.CompleteWithMessages(ctx, userMessages("read"), tool) + if calls.Load() != 1 { + t.Fatalf("second request went to HTTP despite learned local limit: calls=%d", calls.Load()) + } +} + +func TestEqualContextLimitRefreshPersists(t *testing.T) { + t.Cleanup(quirks.resetForTests) + base, model, endpoint := "http://fixture.test", "review/limit", "Wide" + key := contextLimitKey(base, model, endpoint) + old := time.Now().Add(-servingFactHold / 2) + path := filepath.Join(t.TempDir(), quirksFile) + wire := quirksWire{ContextLimits: map[string]ContextLimit{key: {Base: base, Model: model, Provider: endpoint, Tokens: 32768, At: old}}} + encoded, err := json.Marshal(wire) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, encoded, 0o600); err != nil { + t.Fatal(err) + } + quirks.load(path) + client, err := NewClient(Config{BaseURL: base, Model: model, Direct: true}) + if err != nil { + t.Fatal(err) + } + failure := &APIError{Overflow: true, ContextLimit: 32768, Provider: endpoint} + client.rememberContextLimit(model, failure) + if failure.BudgetChanged { + t.Fatal("equal limit was marked as a new recovery budget") + } + quirks.settle() + bytes, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + var onDisk quirksWire + if err := json.Unmarshal(bytes, &onDisk); err != nil { + t.Fatal(err) + } + if !onDisk.ContextLimits[key].At.After(old.Add(time.Minute)) { + t.Fatalf("equal limit date on disk = %s", onDisk.ContextLimits[key].At) + } +} + +func TestFutureDatedContextLimitClampsOnLoadAndStore(t *testing.T) { + t.Cleanup(quirks.resetForTests) + base, model, endpoint := "http://fixture.test", "review/future-limit", "Wide" + key := contextLimitKey(base, model, endpoint) + future := time.Now().Add(24 * time.Hour) + path := filepath.Join(t.TempDir(), quirksFile) + wire := quirksWire{ContextLimits: map[string]ContextLimit{key: {Base: base, Model: model, Provider: endpoint, Tokens: 32768, At: future}}} + encoded, err := json.Marshal(wire) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, encoded, 0o600); err != nil { + t.Fatal(err) + } + quirks.load(path) + if got := quirks.contextLimits[key].At; got.After(time.Now()) { + t.Fatalf("loaded future date = %s", got) + } + quirks.settle() + bytes, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + var repaired quirksWire + if err := json.Unmarshal(bytes, &repaired); err != nil { + t.Fatal(err) + } + if got := repaired.ContextLimits[key].At; got.After(time.Now()) { + t.Fatalf("future date survived on disk = %s", got) + } + storeContextLimit(ContextLimit{Base: base, Model: model, Provider: endpoint, Tokens: 32768, At: future}) + if got := quirks.contextLimits[key].At; got.After(time.Now()) { + t.Fatalf("stored future date = %s", got) + } +} + +func TestUnboundDefaultCeilingIsOmitted(t *testing.T) { + client, err := NewClient(Config{BaseURL: "http://budget.test", Model: "budget/default", Direct: true}) + if err != nil { + t.Fatal(err) + } + for _, tc := range []struct { + prompt int + wantCap bool + }{{60000, false}, {100000, true}} { + request := &ai.Request{Model: "budget/default", Messages: userMessages("read")} + body, err := client.encodeRequest(request, callKnobs{contextBudget: ContextBudget{Window: 131072, Reserve: 65536, PromptFloor: tc.prompt}}) + if err != nil { + t.Fatal(err) + } + var wire map[string]json.RawMessage + if err := json.Unmarshal(body, &wire); err != nil { + t.Fatal(err) + } + _, capped := wire["max_tokens"] + if capped != tc.wantCap { + t.Fatalf("prompt %d max_tokens present=%v, want %v", tc.prompt, capped, tc.wantCap) + } + } +} diff --git a/internal/provider/contextbudget_real_test.go b/internal/provider/contextbudget_real_test.go new file mode 100644 index 0000000000..8b8a00b35a --- /dev/null +++ b/internal/provider/contextbudget_real_test.go @@ -0,0 +1,79 @@ +//go:build e2e + +package provider + +import ( + "context" + "encoding/json" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/home" + lanes "github.com/Agent-Field/codeaf/internal/lane" +) + +// TestRealSheetFitsTheFirstMessageThatWasRefused replays, against the router's +// real endpoints page, the request refused on 2026-09-28: deepseek-v3.2 at +// xhigh, the belt's tools, and about 19,370 tokens of first message and system +// page. The page is public, so no key is needed and nothing is sent to a model: +// the refusal was local, and the encode is the whole of what is being asked. +// +// It asserts the window the request is measured against is one a tool-carrying +// request can reach, and that the full xhigh budget travels. +func TestRealSheetFitsTheFirstMessageThatWasRefused(t *testing.T) { + const model = "deepseek/deepseek-v3.2" + t.Setenv(home.EnvVar, t.TempDir()) + lanes.Default().Reset() + t.Cleanup(lanes.Default().Reset) + client, err := NewClient(Config{APIKey: "sk-no-key-needed-for-the-endpoints-page", BaseURL: "https://openrouter.ai/api/v1", Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Skipf("the router's endpoints page could not be read: %v", err) + } + rows := lanes.Default().Sheet().Rows(model) + smallest, smallestWithTools := 0, 0 + for _, row := range rows { + t.Logf("%-14s context %7d tools %v", row.ID.Lane, row.Facts.Context, row.Facts.Tools) + if c := row.Facts.Context; c > 0 && (smallest == 0 || c < smallest) { + smallest = c + } + if c := row.Facts.Context; c > 0 && row.Facts.Tools && (smallestWithTools == 0 || c < smallestWithTools) { + smallestWithTools = c + } + } + window := client.servingWindow(model, nil, 163840, true) + t.Logf("smallest window %d, smallest taking tools %d, measured against %d", smallest, smallestWithTools, window) + if window != min(163840, smallestWithTools) { + t.Fatalf("a tool-carrying request was measured against %d, want the smallest tool endpoint's %d", window, smallestWithTools) + } + + request := &ai.Request{ + Model: model, + Messages: userMessages(strings.Repeat("x", 19370*4)), + Tools: []ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read", Parameters: map[string]any{"type": "object"}}}}, + } + knobs := callKnobs{ + contextBudget: ContextBudget{Window: 163840, Reserve: 16384}, + effort: effortRequest{effort: EffortHigh, budget: xhighReasoningTokens, explicit: true}, + } + body, err := client.encodeRequest(request, knobs) + if err != nil { + t.Fatalf("the refused first message is still refused: %v", err) + } + var wire struct { + MaxTokens int `json:"max_tokens"` + Reasoning struct { + MaxTokens int `json:"max_tokens"` + } `json:"reasoning"` + } + if err := json.Unmarshal(body, &wire); err != nil { + t.Fatal(err) + } + t.Logf("sent max_tokens %d, reasoning.max_tokens %d", wire.MaxTokens, wire.Reasoning.MaxTokens) + if wire.Reasoning.MaxTokens != xhighReasoningTokens { + t.Fatalf("thinking budget %d, want the whole xhigh %d on a window this large", wire.Reasoning.MaxTokens, xhighReasoningTokens) + } +} diff --git a/internal/provider/contextbudget_relay_contract_test.go b/internal/provider/contextbudget_relay_contract_test.go new file mode 100644 index 0000000000..39162ddebb --- /dev/null +++ b/internal/provider/contextbudget_relay_contract_test.go @@ -0,0 +1,167 @@ +package provider + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + lanes "github.com/Agent-Field/codeaf/internal/lane" +) + +func TestRelayedSmallToollessWindowGetsOneResendWithoutCappingTools(t *testing.T) { + forgetLanes(t) + model := "review/windows" + mara, err := os.ReadFile("testdata/mara-overflow.json") + if err != nil { + t.Fatal(err) + } + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + if strings.HasSuffix(r.URL.Path, "/endpoints") { + _, _ = w.Write([]byte(`{"data":{"endpoints":[{"provider_name":"Mara","context_length":32768,"max_completion_tokens":8192,"supported_parameters":[]},{"provider_name":"Wide","context_length":131072,"max_completion_tokens":8192,"supported_parameters":["tools"]}]}}`)) + return + } + if calls.Add(1) == 1 { + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write(mara) + return + } + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"answered"},"finish_reason":"stop"}]}`)) + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 131072, Reserve: 8192, PromptFloor: 40000}) + tools := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + for i := 0; i < 2; i++ { + answer, err := client.CompleteWithMessages(ctx, userMessages("read"), tools) + if err != nil || answer == nil || answer.Text() != "answered" { + t.Fatalf("tool request %d: answer=%v error=%v calls=%d", i+1, answer, err, calls.Load()) + } + } + if got := calls.Load(); got != 3 { + t.Fatalf("tool requests made %d HTTP calls, want the first retried once and the next sent once", got) + } + if got := client.servingWindow(model, nil, 131072, true); got != 131072 { + t.Fatalf("tool request inherited Mara's learned window: %d", got) + } + _, err = client.CompleteWithMessages(ctx, userMessages("plain")) + failure, ok := RefusalFrom(err) + if !ok || !failure.Local || !failure.Overflow || calls.Load() != 3 { + t.Fatalf("no-tools request: error=%v calls=%d; want local small-window refusal", err, calls.Load()) + } +} + +func TestRelayedSmallWindowStopsAfterOneResend(t *testing.T) { + mara, err := os.ReadFile("testdata/mara-overflow.json") + if err != nil { + t.Fatal(err) + } + // A router can answer a resend from another endpoint, so it gets exactly + // one; a direct base is the one endpoint, so resending it the same request + // would only pay the refusal twice. + for _, road := range []struct { + name string + direct bool + want int32 + }{{"router", false, 2}, {"direct", true, 1}} { + t.Run(road.name, func(t *testing.T) { + forgetLanes(t) + model := "review/repeated-" + road.name + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + if strings.HasSuffix(r.URL.Path, "/endpoints") { + _, _ = w.Write([]byte(`{"data":{"endpoints":[{"provider_name":"Mara","context_length":32768,"supported_parameters":[]},{"provider_name":"Wide","context_length":131072,"supported_parameters":["tools"]}]}}`)) + return + } + calls.Add(1) + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write(mara) + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: model, Direct: road.direct}) + if err != nil { + t.Fatal(err) + } + if !road.direct { + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + } + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 131072, Reserve: 8192, PromptFloor: 40000}) + _, err = client.CompleteWithMessages(ctx, userMessages("read"), ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}})) + failure, ok := RefusalFrom(err) + if !ok || !failure.Overflow || calls.Load() != road.want { + t.Fatalf("overflow: error=%v requests=%d, want the overflow after %d request(s)", err, calls.Load(), road.want) + } + }) + } +} + +func TestContextLimitExpiresOnLoadAndFreshToolLimitStillApplies(t *testing.T) { + forgetLanes(t) + t.Cleanup(quirks.resetForTests) + model := "review/ttl" + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if strings.HasSuffix(r.URL.Path, "/endpoints") { + _, _ = w.Write([]byte(`{"data":{"endpoints":[{"provider_name":"Wide","context_length":131072,"supported_parameters":["tools"]}]}}`)) + } + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + path := filepath.Join(t.TempDir(), quirksFile) + key := contextLimitKey(server.URL, model, "Wide") + encoded, _ := json.Marshal(quirksWire{ContextLimits: map[string]ContextLimit{ + key: {Base: server.URL, Model: model, Provider: "Wide", Tokens: 32768}, + }}) + if err := os.WriteFile(path, encoded, 0o600); err != nil { + t.Fatal(err) + } + quirks.load(path) + if got := client.servingWindow(model, nil, 131072, true); got != 131072 { + t.Fatalf("undated persisted limit survived load: %d", got) + } + storeContextLimit(ContextLimit{Base: server.URL, Model: model, Provider: "Wide", Tokens: 65536}) + if got := client.servingWindow(model, nil, 131072, true); got != 65536 { + t.Fatalf("fresh tool endpoint limit = %d, want 65536", got) + } + storeContextLimit(ContextLimit{Base: server.URL, Model: model, Provider: "Unlisted", Tokens: 49152}) + if got := client.servingWindow(model, nil, 131072, true); got != 49152 { + t.Fatalf("unknown endpoint limit = %d, want conservative 49152", got) + } + quirks.forget() + stale, err := json.Marshal(quirksWire{ContextLimits: map[string]ContextLimit{ + key: {Base: server.URL, Model: model, Provider: "Wide", Tokens: 32768, At: time.Now().Add(-time.Hour)}, + }}) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, stale, 0o600); err != nil { + t.Fatal(err) + } + quirks.load(path) + if got := client.servingWindow(model, nil, 131072, true); got != 131072 { + t.Fatalf("expired dated limit = %d, want catalog window", got) + } +} diff --git a/internal/provider/contextbudget_test.go b/internal/provider/contextbudget_test.go new file mode 100644 index 0000000000..08762351de --- /dev/null +++ b/internal/provider/contextbudget_test.go @@ -0,0 +1,251 @@ +package provider + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + lanes "github.com/Agent-Field/codeaf/internal/lane" + "github.com/Agent-Field/codeaf/internal/lane/lanestub" +) + +func TestContextBudgetReadsTheReportedTotalNotThePrompt(t *testing.T) { + body := `{"error":{"message":"Upstream error from DeepInfra: This model's maximum context length is 40960 tokens. However, you requested 14746 output tokens and your prompt contains at least 26215 input tokens, for a total of at least 40961 tokens.","metadata":{"provider_name":"DeepInfra"}}}` + failure, _ := RefusalFrom(apiError(400, []byte(body))) + if !failure.Overflow || failure.ContextLimit != 40960 || failure.InputTokens != 26215 || failure.OutputTokens != 14746 { + t.Fatalf("wrong evidence: %+v", failure) + } + other, _ := RefusalFrom(apiError(400, []byte(`{"error":{"message":"context length exceeded by 12 tokens"}}`))) + if other.ContextLimit != 0 { + t.Fatalf("invented total limit: %+v", other) + } +} + +func TestContextBudgetCountsToolsAndReplayedReasoning(t *testing.T) { + client, _ := NewClient(Config{BaseURL: "http://budget.test", Model: "budget/count", Direct: true}) + request := &ai.Request{Model: "budget/count", Messages: userMessages("hello")} + knobs := callKnobs{contextBudget: ContextBudget{Window: 8192, Reserve: 2048}} + if _, err := client.encodeRequest(request, knobs); err != nil { + t.Fatal(err) + } + request.Tools = []ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "big", Parameters: map[string]any{"description": strings.Repeat("schema ", 6000)}}}} + if _, err := client.encodeRequest(request, knobs); err == nil { + t.Fatal("oversized tool schema was not counted") + } + request.Tools = nil + request.Messages = append(request.Messages, ai.Message{Role: "assistant", Content: []ai.ContentPart{{Type: "text", Text: "working"}}}) + knobs.reasoning = []MessageReasoning{{}, {Field: "reasoning", Text: strings.Repeat("working ", 5000), Model: request.Model}} + if _, err := client.encodeRequest(request, knobs); err == nil { + t.Fatal("replayed reasoning was not counted") + } +} + +func TestContextBudgetStopsOversizeBeforeHTTP(t *testing.T) { + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { calls.Add(1); http.Error(w, "must not send", 500) })) + defer server.Close() + client, _ := NewClient(Config{APIKey: "test", BaseURL: server.URL, Model: "budget/oversize", Direct: true}) + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 8192, Reserve: 2048}) + _, err := client.CompleteWithMessages(ctx, userMessages(strings.Repeat("input ", 9000))) + failure, ok := RefusalFrom(err) + if !ok || !failure.Local || !failure.Overflow || calls.Load() != 0 { + t.Fatalf("error %v, requests %d", err, calls.Load()) + } +} + +func TestContextBudgetFitsInputAndOutputAndIgnoresOldModelMemo(t *testing.T) { + model := "budget/fit" + quirksAt(t, model) + NoteServedWindow(model, 26217) + client, _ := NewClient(Config{BaseURL: "http://budget.test", Model: model, Direct: true}) + failure := &APIError{Overflow: true, ContextLimit: 40960, Provider: "DeepInfra"} + client.rememberContextLimit(model, failure) + request := &ai.Request{Model: model, Messages: userMessages(strings.Repeat("x", 26215*4))} + body, err := client.encodeRequest(request, callKnobs{contextBudget: ContextBudget{Window: 131072, Reserve: 65536}}) + if err != nil { + t.Fatal(err) + } + var wire struct { + MaxTokens int `json:"max_tokens"` + } + if err = json.Unmarshal(body, &wire); err != nil { + t.Fatal(err) + } + if wire.MaxTokens <= 0 || 26215+wire.MaxTokens+ContextSafetyTokens(40960) > 40960 { + t.Fatalf("output allowance %d does not fit", wire.MaxTokens) + } + // A pinned different endpoint and a different account do not inherit this + // endpoint's refusal. An advisory order may still land on the smaller one. + if got := client.servingWindow(model, &providerPrefs{Only: []string{"Other"}}, 131072, false); got != 131072 { + t.Fatalf("other endpoint got %d", got) + } + other, _ := NewClient(Config{BaseURL: "http://another.test", Model: model, Direct: true}) + if got := other.servingWindow(model, nil, 131072, false); got != 131072 { + t.Fatalf("other account got %d", got) + } + quirks.settle() + path, _ := quirks.snapshot() + fresh := &quirksStore{mandatory: map[string]time.Time{}, disableIgnored: map[string]time.Time{}, noCacheControl: map[string]time.Time{}, noReasoningBudget: map[string]time.Time{}, noReasoningReplay: map[string]time.Time{}, answerCut: map[string]int{}, servedWindow: map[string]int{}} + fresh.load(path) + key := contextLimitKey(client.config.BaseURL, model, "DeepInfra") + if got := fresh.contextLimits[key].Tokens; got != 40960 { + t.Fatalf("reloaded context limit = %d", got) + } +} + +func TestContextBudgetAllowsShorterAnswerOnSmallWindow(t *testing.T) { + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body struct { + MaxTokens int `json:"max_tokens"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Error(err) + } + calls.Add(1) + // The reported lean launch leaves 1,663 tokens after its safety margin. + if body.MaxTokens != 1663 { + t.Errorf("output allowance = %d, want the available 1663", body.MaxTokens) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"A short answer fits."},"finish_reason":"stop"}]}`)) + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: "budget/small-answer", Direct: true}) + if err != nil { + t.Fatal(err) + } + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 16385, Reserve: 4096, PromptFloor: 13903}) + response, err := client.CompleteWithMessages(ctx, userMessages("Explain the conjecture.")) + if err != nil || response == nil || calls.Load() != 1 { + t.Fatalf("response=%v error=%v HTTP requests=%d", response, err, calls.Load()) + } +} + +func TestContextBudgetStillRequiresUsefulAnswerRoom(t *testing.T) { + client, _ := NewClient(Config{BaseURL: "http://budget.test", Model: "budget/minimum", Direct: true}) + request := &ai.Request{Model: "budget/minimum", Messages: userMessages("hello")} + for _, test := range []struct { + name string + room int + wantError bool + }{ + {"short answer fits", minimumContextAnswer, false}, + {"not enough room", minimumContextAnswer - 1, true}, + } { + t.Run(test.name, func(t *testing.T) { + knobs := callKnobs{contextBudget: ContextBudget{Window: 16385, Reserve: 4096, PromptFloor: 16385 - ContextSafetyTokens(16385) - test.room}} + body, err := client.encodeRequest(request, knobs) + if (err != nil) != test.wantError { + t.Fatalf("encode error=%v", err) + } + if err == nil { + var wire struct { + MaxTokens int `json:"max_tokens"` + } + if err := json.Unmarshal(body, &wire); err != nil { + t.Fatal(err) + } + if wire.MaxTokens != test.room { + t.Fatalf("allowance=%d room=%d", wire.MaxTokens, test.room) + } + } + }) + } +} + +// THE THINKING BUDGET BENDS TO THE WINDOW. The xhigh rung asks for 32,000 +// tokens of thinking; on a 32k window that alone is larger than the window, and +// reserving it refused a first message with nothing to compact (2026-09-28). +// The budget now shrinks to what the prompt leaves, the output ceiling still +// holds it and an answer, and the whole request fits. +func TestAThinkingBudgetShrinksToTheWindowInsteadOfRefusing(t *testing.T) { + client, _ := NewClient(Config{BaseURL: "http://budget.test", Model: "budget/thinks", Direct: true, + SupportsParameter: func(string, string) (bool, bool) { return true, true }}) + const window = 32768 + request := &ai.Request{Model: "budget/thinks", Messages: userMessages(strings.Repeat("x", 19370*4))} + knobs := callKnobs{ + contextBudget: ContextBudget{Window: window, Reserve: 8192}, + effort: effortRequest{effort: EffortHigh, budget: xhighReasoningTokens, explicit: true}, + } + body, err := client.encodeRequest(request, knobs) + if err != nil { + t.Fatalf("a first message on a 32k window was refused: %v", err) + } + var wire struct { + MaxTokens int `json:"max_tokens"` + Reasoning struct { + MaxTokens int `json:"max_tokens"` + } `json:"reasoning"` + } + if err = json.Unmarshal(body, &wire); err != nil { + t.Fatal(err) + } + thinking := wire.Reasoning.MaxTokens + if thinking < minimumContextThinking || thinking >= xhighReasoningTokens { + t.Fatalf("thinking budget %d, want it shrunk below %d and kept above %d", thinking, xhighReasoningTokens, minimumContextThinking) + } + if wire.MaxTokens <= thinking { + t.Fatalf("output ceiling %d leaves no answer after %d of thinking", wire.MaxTokens, thinking) + } + if 19370+wire.MaxTokens+ContextSafetyTokens(window) > window { + t.Fatalf("output ceiling %d does not fit behind the prompt", wire.MaxTokens) + } + + // With too little room for a budget worth sending, the budget is dropped + // and the effort travels as its word, still under a ceiling that fits. + request.Messages = userMessages(strings.Repeat("x", 29500*4)) + body, err = client.encodeRequest(request, knobs) + if err != nil { + t.Fatalf("a prompt with room for a short answer was refused: %v", err) + } + var word struct { + MaxTokens int `json:"max_tokens"` + Reasoning struct { + Effort string `json:"effort"` + MaxTokens int `json:"max_tokens"` + } `json:"reasoning"` + } + if err = json.Unmarshal(body, &word); err != nil { + t.Fatal(err) + } + if word.Reasoning.MaxTokens != 0 || word.Reasoning.Effort != string(EffortHigh) { + t.Fatalf("reasoning = %+v, want the word without a budget", word.Reasoning) + } + if 29500+word.MaxTokens+ContextSafetyTokens(window) > window { + t.Fatalf("output ceiling %d does not fit behind the prompt", word.MaxTokens) + } +} + +// A REQUEST CARRYING TOOLS IS MEASURED AGAINST THE ENDPOINTS THAT TAKE THEM. +// deepseek-v3.2's sheet on 2026-09-28 had two 32k endpoints that take no +// tools; counting them made every tool-carrying request fit a 32k window that +// no such request could ever be sent to. +func TestTheWindowIgnoresEndpointsThatCannotTakeTheRequestsTools(t *testing.T) { + forgetLanes(t) + const model = "openrouter/windows" + server := lanestub.New(model, + lanestub.Lane{Name: "wide", Profile: lanestub.Profile{TTFT: 20 * time.Millisecond, Rate: 400, Tokens: 8, Tools: true, Context: 131072}}, + lanestub.Lane{Name: "narrow", Profile: lanestub.Profile{TTFT: 20 * time.Millisecond, Rate: 400, Tokens: 8, Tools: false, Context: 32768}}, + ) + t.Cleanup(server.Close) + client, err := NewClient(Config{APIKey: "test-key", BaseURL: server.URL(), Model: model}) + if err != nil { + t.Fatal(err) + } + if err := lanes.Default().Sheet().Refresh(context.Background(), model); err != nil { + t.Fatal(err) + } + if got := client.servingWindow(model, nil, 163840, true); got != 131072 { + t.Fatalf("a request carrying tools measured against %d, want the tool endpoint's 131072", got) + } + if got := client.servingWindow(model, nil, 163840, false); got != 32768 { + t.Fatalf("a request without tools measured against %d, want the smallest endpoint's 32768", got) + } +} diff --git a/internal/provider/deepinfra_overflow_contract_test.go b/internal/provider/deepinfra_overflow_contract_test.go new file mode 100644 index 0000000000..bbda506d77 --- /dev/null +++ b/internal/provider/deepinfra_overflow_contract_test.go @@ -0,0 +1,32 @@ +package provider + +import ( + "testing" + + "github.com/Agent-Field/codeaf/internal/taxonomy" +) + +func TestDeepInfraInputLengthRefusalCarriesItsExactLimits(t *testing.T) { + body := []byte(`{"error":{"message":"Provider returned error","code":400,"metadata":{"raw":"Requested input length 39818 exceeds maximum input length 32767 (via DeepInfra)","provider_name":"DeepInfra"}}}`) + failure, ok := RefusalFrom(apiError(400, body)) + if !ok || !failure.Overflow || failure.ContextLimit != 32767 || failure.InputTokens != 39818 || failure.Provider != "DeepInfra" { + t.Fatalf("DeepInfra overflow evidence = %+v, ok %v", failure, ok) + } + if verdict := taxonomy.Classify(Evidence(failure), taxonomy.Limits{}.Floored()); !verdict.Compacts() { + t.Fatalf("DeepInfra overflow chose %s instead of recovery", verdict) + } + model := "review/deepinfra-overflow" + quirksAt(t, model) + client, err := NewClient(Config{BaseURL: "http://deepinfra.test", Model: model, Direct: true}) + if err != nil { + t.Fatal(err) + } + client.rememberContextLimit(model, failure) + if got := client.servingWindow(model, nil, 100_000, false); got != 32767 { + t.Fatalf("learned endpoint limit = %d, want 32767", got) + } + other, _ := RefusalFrom(apiError(400, []byte(`{"error":{"message":"Provider returned error","code":400,"metadata":{"raw":"The input file could not be read","provider_name":"DeepInfra"}}}`))) + if other.Overflow { + t.Fatalf("unrelated refusal became overflow: %+v", other) + } +} diff --git a/internal/provider/dispatch.go b/internal/provider/dispatch.go index 89032c2424..dee4c8c4ee 100644 --- a/internal/provider/dispatch.go +++ b/internal/provider/dispatch.go @@ -794,6 +794,14 @@ func (c *Client) send(ctx context.Context, request *ai.Request, knobs callKnobs, if peek == nil { peek, _ = io.ReadAll(io.LimitReader(response.Body, maxErrorPeek)) } + // A context refusal needs a smaller request, even when the router names + // the machine that refused it. Return it to the conversation before an + // endpoint walk can resend the same oversized bytes. + if failure, ok := RefusalFrom(apiError(response.StatusCode, peek)); ok && failure.Overflow { + response.Body = rewound(peek, response.Body) + sharedLimiter.release(false, 0) + return response, nil + } response.Body.Close() cancelAttempt() lastErr = apiError(response.StatusCode, peek) @@ -1230,6 +1238,14 @@ func (c *Client) recoverFromRefusal( return nil, err } if response != nil { + // THE RUNG THAT LANDED IS REMEMBERED WHEN IT WAS THE TOOLS: every + // cheaper rung was already on and refused, so this model is not + // served with tools here, and the next turn is sent without them + // rather than refused again (toolless.go). + if step.bit == relaxTools && response.StatusCode < 400 { + c.toolless.learn(model) + c.leaveOffTools(ctx, model, relaxed, false) + } return response, nil } last = payload diff --git a/internal/provider/endpoints.go b/internal/provider/endpoints.go index 375d3a1e30..63c5846419 100644 --- a/internal/provider/endpoints.go +++ b/internal/provider/endpoints.go @@ -353,6 +353,7 @@ var overflowPhrases = []string{ "context limit", "prompt is too long", "too many tokens", + "exceeds maximum input length", } // overflowRefusal reports that a refusal is the request not fitting: the @@ -493,7 +494,7 @@ var relaxRungs = []relaxStep{ {bit: relaxMaxTokens, label: "removed max_tokens", name: "max_tokens"}, {bit: relaxResponseFormat, label: "removed response_format", name: "response_format"}, {bit: relaxImages, label: "removed images", name: "images"}, - {bit: relaxTools, label: "removed tools", name: "tools"}, + {bit: relaxTools, label: "sent without tools, which no provider serving this model accepts — it cannot read, search or change files on this answer", name: "tools"}, } // widenedOff is what the first rung really takes off ONE preference object: @@ -591,7 +592,7 @@ func (c *Client) relaxationPlan(request *ai.Request, knobs callKnobs, model stri if carriesAttachments(request.Messages) { plan = append(plan, rung(relaxImages)) } - if len(request.Tools) > 0 { + if len(request.Tools) > 0 && !knobs.relaxed.has(relaxTools) { plan = append(plan, rung(relaxTools)) } return plan diff --git a/internal/provider/endpoints_test.go b/internal/provider/endpoints_test.go index e327f8a876..c05cce20e7 100644 --- a/internal/provider/endpoints_test.go +++ b/internal/provider/endpoints_test.go @@ -85,7 +85,7 @@ func TestRefusalStripsOptionalParametersInOrderAndSaysSo(t *testing.T) { "Retry 1/4: relaxed the endpoint filter", "Retry 2/4: removed reasoning", "Retry 3/4: removed max_tokens", - "Retry 4/4: removed tools", + "Retry 4/4: sent without tools, which no provider serving this model accepts — it cannot read, search or change files on this answer", } if len(notices) != len(want) { t.Fatalf("notices = %#v, want %d lines", notices, len(want)) diff --git a/internal/provider/foldedladder_test.go b/internal/provider/foldedladder_test.go index 5a975101f4..08cafd5d35 100644 --- a/internal/provider/foldedladder_test.go +++ b/internal/provider/foldedladder_test.go @@ -61,7 +61,7 @@ func TestTheLadderTakesItsRungFromTheMoveGenerator(t *testing.T) { want := []string{ "Retry 1/3: relaxed the endpoint filter", "Retry 2/3: removed max_tokens", - "Retry 3/3: removed tools", + "Retry 3/3: sent without tools, which no provider serving this model accepts — it cannot read, search or change files on this answer", } if len(notices) != len(want) { t.Fatalf("the ladder said %#v, want %#v", notices, want) diff --git a/internal/provider/quirks.go b/internal/provider/quirks.go index bf3904ebb5..adb07f0d2d 100644 --- a/internal/provider/quirks.go +++ b/internal/provider/quirks.go @@ -76,29 +76,13 @@ type quirksStore struct { // one-object verdict, and a memo that pooled them would raise the ceiling // on every call in the system because one of them is wide. answerCut map[string]int - // servedWindow is the seventh learned fact and the second that is a NUMBER: - // the largest prompt, in tokens, that a model was REFUSED for being too - // long, keyed by model alone. - // - // It is the same kind of fact as its neighbours — something a provider will - // not publish and only a call's answer can teach — and it exists because the - // published figure is sometimes wrong by an order of magnitude. The catalog - // row for ~deepseek/deepseek-v4-flash-latest claims 1,310,720 tokens; the - // endpoint serving it did not serve anything like that, and the compaction - // law believed the row. What an overflow refusal says is not a claim but a - // measurement: THIS many tokens was too many, here, today. - // - // It is keyed by model and not by lane, which is the one place it differs - // from answerCut above. A window is a property of the model's serving - // weights rather than of one replica's completion budget, and a memo that - // re-learned it per endpoint would spend one over-long request per endpoint - // discovering the same fact. - // - // It only ever SHRINKS, and it is written down for the same reason every - // other fact here is: so that the next process does not have to be told - // again. - servedWindow map[string]int - loaded bool + // servedWindow retains the legacy on-disk field for compatibility. Those + // values were rejected prompt estimates, not total context limits, and + // request sizing no longer reads them. Explicit endpoint limits live in + // contextLimits, keyed by base URL, model and serving endpoint. + servedWindow map[string]int + contextLimits map[string]ContextLimit + loaded bool // writes counts saves in flight. The save is deliberately off the request // path — the call that learned the fact is waiting to be re-sent and must @@ -177,7 +161,8 @@ type quirksWire struct { // ServedWindow maps a model to the largest prompt, in tokens, it has been // refused for. It only ever shrinks, and it is read as a CEILING on what the // catalog claims rather than as a window in its own right. - ServedWindow map[string]int `json:"served_window,omitempty"` + ServedWindow map[string]int `json:"served_window,omitempty"` + ContextLimits map[string]ContextLimit `json:"context_limits,omitempty"` } func (q *quirksStore) load(path string) { @@ -193,6 +178,21 @@ func (q *quirksStore) load(path string) { if json.Unmarshal(raw, &wire) != nil { return } + if q.contextLimits == nil { + q.contextLimits = make(map[string]ContextLimit) + } + clamped := false + for key, limit := range wire.ContextLimits { + // Older builds wrote no date. Such a limit has no evidence that the + // endpoint still has that window, so a restart lets it lapse. + if limit.Tokens > 0 && !limit.At.IsZero() && limit.At.After(time.Now().Add(-servingFactHold)) { + if now := time.Now(); limit.At.After(now) { + limit.At = now + clamped = true + } + q.contextLimits[key] = limit + } + } seed(q.mandatory, wire.ReasoningMandatory) seed(q.disableIgnored, wire.ReasoningDisableIgnored) seed(q.noCacheControl, wire.CacheControlRejected) @@ -213,6 +213,11 @@ func (q *quirksStore) load(path string) { } } } + if clamped { + // A future date must be repaired on disk too, or each restart gives + // that stale refusal a fresh hold again. + q.persist() + } } // seed folds a loaded set into a live one without ever dropping a fact learned @@ -428,6 +433,10 @@ func (q *quirksStore) snapshot() (string, quirksWire) { ReasoningReplayRejected: make(map[string]time.Time, len(q.noReasoningReplay)), AnswerCutAt: make(map[string]int, len(q.answerCut)), ServedWindow: make(map[string]int, len(q.servedWindow)), + ContextLimits: make(map[string]ContextLimit, len(q.contextLimits)), + } + for key, limit := range q.contextLimits { + wire.ContextLimits[key] = limit } for key, spent := range q.answerCut { wire.AnswerCutAt[key] = spent @@ -597,6 +606,7 @@ func (q *quirksStore) forget() { q.noReasoningReplay = map[string]time.Time{} q.answerCut = map[string]int{} q.servedWindow = map[string]int{} + q.contextLimits = map[string]ContextLimit{} q.path = "" q.loaded = false } diff --git a/internal/provider/quirks_test.go b/internal/provider/quirks_test.go index 895ba596f5..1818b6b390 100644 --- a/internal/provider/quirks_test.go +++ b/internal/provider/quirks_test.go @@ -200,6 +200,13 @@ func quirksAt(t *testing.T, models ...string) string { quirks.settle() quirks.mutex.Lock() defer quirks.mutex.Unlock() + for key, limit := range quirks.contextLimits { + for _, model := range models { + if limit.Model == normalizeModel(model) { + delete(quirks.contextLimits, key) + } + } + } for _, model := range models { delete(quirks.mandatory, normalizeModel(model)) delete(quirks.disableIgnored, normalizeModel(model)) diff --git a/internal/provider/testdata/mara-overflow.json b/internal/provider/testdata/mara-overflow.json new file mode 100644 index 0000000000..f5eac71aa6 --- /dev/null +++ b/internal/provider/testdata/mara-overflow.json @@ -0,0 +1 @@ +{"error":{"message":"Provider returned error","code":400,"metadata":{"raw":"{\"error\":{\"code\":\"context_length_exceeded\",\"message\":\"This model's maximum context length is 32768 tokens. However, your messages resulted in 40283 tokens. Please reduce the length of the messages.\",\"param\":\"messages\",\"type\":\"invalid_request_error\"},\"request_id\":\"datdej710qhmd1fokr70\"}\n","provider_name":"Mara","is_byok":false,"provider_error_code":"context_length_exceeded"}},"user_id":"<redacted>"} diff --git a/internal/provider/thinking.go b/internal/provider/thinking.go index 3d9d9921ff..407d7deb6c 100644 --- a/internal/provider/thinking.go +++ b/internal/provider/thinking.go @@ -120,6 +120,10 @@ func (c *Client) lowestEffort(model string) Effort { if !known || len(profile.Efforts) == 0 { return EffortLow } + return lowestListedEffort(profile) +} + +func lowestListedEffort(profile ReasoningProfile) Effort { lowest := EffortNone for _, word := range profile.Efforts { if word == EffortNone || effortRank(word) >= effortRank("") { @@ -135,11 +139,32 @@ func (c *Client) lowestEffort(model string) Effort { return lowest } +// listedEffort raises a requested word to the catalog's lowest usable level. +// A model listing only high and xhigh cannot run an optional low pass at low. +func listedEffort(profile ReasoningProfile, requested Effort) Effort { + if len(profile.Efforts) == 0 || requested == EffortNone || requested == EffortOff { + return requested + } + floor := lowestListedEffort(profile) + if effortRank(requested) < effortRank(floor) { + return floor + } + return requested +} + // runningEffort is the level the thinking pass will actually run at: the word // being sent, or — when nothing is sent to a model that thinks regardless — // the row's default, taken as the top of the ladder when the row does not say. func (c *Client) runningEffort(model string, sent Effort) Effort { if sent != EffortNone && sent != EffortOff { + // A MODEL RUNS NO LOWER THAN ITS PUBLISHED FLOOR. The word travels as + // asked, but a model whose catalog lists only high and xhigh thinks at + // one of those whatever it is sent, and a ceiling sized for low left a + // summary no room to answer (deepseek-v4-flash, 2026-09-28: 8 of 28 + // summaries ended at the ceiling with no text). + if profile, known := c.profileFor(model); known { + return listedEffort(profile, sent) + } return sent } if !c.reasoningUnstoppable(model) { @@ -185,17 +210,35 @@ func (c *Client) wireCeiling(model string, sent Effort, budget int, answer int) return answer + budget } running := c.runningEffort(model, sent) - share := thinkingShare(running) - if share <= 0 { + if thinkingShare(running) <= 0 { return answer } - ceiling := int(float64(answer)/(1-share) + 0.5) + ceiling := reasoningCeiling(answer, running) if running != sent && ceiling < answer+unaskedThinkingFloor { return answer + unaskedThinkingFloor } return ceiling } +// SummaryOutputReserve is the space a summary chunk must leave behind its +// prompt when the catalog publishes a floor above the summarizer's low ask. +// The provider's wire sizing uses the same reasoningCeiling calculation. +func SummaryOutputReserve(profile ReasoningProfile, answer int) int { + floor := listedEffort(profile, EffortLow) + if effortRank(floor) <= effortRank(EffortLow) { + return answer + } + return reasoningCeiling(answer, floor) +} + +func reasoningCeiling(answer int, running Effort) int { + share := thinkingShare(running) + if share <= 0 { + return answer + } + return int(float64(answer)/(1-share) + 0.5) +} + // ceilingFor is the wire ceiling for one request as its knobs will shape it, // and whether the request carries a ceiling at all. It is read by the encoder // and by the transport, which is the point: the wait is sized for the reply diff --git a/internal/provider/thinking_summary_contract_test.go b/internal/provider/thinking_summary_contract_test.go new file mode 100644 index 0000000000..f74f03c043 --- /dev/null +++ b/internal/provider/thinking_summary_contract_test.go @@ -0,0 +1,38 @@ +package provider + +import ( + "context" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestListedThinkingFloorLeavesAnswerRoomInsideWindow(t *testing.T) { + const model = "listed/summary-thinker" + client, recorded := newTestClient(t, Config{ + Model: model, + SupportsParameter: func(string, string) (bool, bool) { return true, true }, + ReasoningProfile: func(string) (ReasoningProfile, bool) { + return ReasoningProfile{Mandatory: true, Efforts: []Effort{"xhigh", EffortHigh}, Default: "xhigh"}, true + }, + }) + ctx := WithContextBudget(WithReasoningEffort(context.Background(), EffortLow), ContextBudget{Window: 8192, Reserve: 1000}) + if _, err := client.CompleteWithMessages(ctx, userMessages("Summarize this conversation"), ai.WithMaxTokens(1000)); err != nil { + t.Fatal(err) + } + if recorded.count() != 1 { + t.Fatalf("requests = %d, want one", recorded.count()) + } + body := recorded.body(0) + reasoning, _ := body["reasoning"].(map[string]any) + // The word travels as asked; only the room is sized for the level the + // model can actually run at. + if reasoning["effort"] != "low" { + t.Fatalf("reasoning = %#v, want the requested effort word unchanged", reasoning) + } + // The model runs at high at the least, which thinks with 80% of the + // ceiling: 1,000 tokens of answer need 5,000, and the window holds them. + if ceiling, _ := body["max_tokens"].(float64); ceiling < 5000 || ceiling >= 8192 { + t.Fatalf("max_tokens = %v, want room for a high thinking pass above the 1,000-token answer, inside the window", body["max_tokens"]) + } +} diff --git a/internal/provider/toolhistory.go b/internal/provider/toolhistory.go new file mode 100644 index 0000000000..dc2417b8ef --- /dev/null +++ b/internal/provider/toolhistory.go @@ -0,0 +1,61 @@ +package provider + +import ( + "fmt" + "strings" + "unicode/utf8" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +const toolHistoryArgumentLimit = 512 + +// readableToolHistory carries completed calls as ordinary conversation text +// when an endpoint cannot accept the tool protocol. It leaves every original +// content part in place and only clips long arguments, which are a description +// of an earlier call rather than the result of that call. +func readableToolHistory(messages []ai.Message) []ai.Message { + var rewritten []ai.Message + names := make(map[string]string) + for index, message := range messages { + if len(message.ToolCalls) == 0 && message.Role != "tool" { + continue + } + if rewritten == nil { + rewritten = append([]ai.Message(nil), messages...) + } + message.Content = append([]ai.ContentPart(nil), message.Content...) + if message.Role == "tool" { + name := names[message.ToolCallID] + if name == "" { + name = "earlier call" + } + message.Role = "user" + message.Content = append([]ai.ContentPart{{Type: "text", Text: fmt.Sprintf("Result from %s (%s):\n", name, message.ToolCallID)}}, message.Content...) + message.ToolCallID = "" + } else { + var lines strings.Builder + for _, call := range message.ToolCalls { + names[call.ID] = call.Function.Name + arguments := call.Function.Arguments + // The limit is in bytes, cut back to the start of a character so a + // clipped argument is never a broken one. + if len(arguments) > toolHistoryArgumentLimit { + cut := toolHistoryArgumentLimit + for cut > 0 && !utf8.RuneStart(arguments[cut]) { + cut-- + } + arguments = arguments[:cut] + "…" + } + fmt.Fprintf(&lines, "\nCalled %s (%s) with %s", call.Function.Name, call.ID, arguments) + } + message.Content = append(message.Content, ai.ContentPart{Type: "text", Text: lines.String()}) + message.ToolCalls = nil + } + rewritten[index] = message + } + if rewritten == nil { + return messages + } + return rewritten +} diff --git a/internal/provider/toolhistory_clip_test.go b/internal/provider/toolhistory_clip_test.go new file mode 100644 index 0000000000..ecaa507ab5 --- /dev/null +++ b/internal/provider/toolhistory_clip_test.go @@ -0,0 +1,23 @@ +package provider + +import ( + "strings" + "testing" + "unicode/utf8" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +// A clipped argument is cut at a character boundary at or under the byte +// limit, so the text a no-tools model reads is never a broken character. +func TestToolHistoryClipsArgumentsAtACharacterBoundary(t *testing.T) { + long := `{"path":"` + strings.Repeat("é", 400) + `"}` + messages := readableToolHistory([]ai.Message{{Role: "assistant", ToolCalls: []ai.ToolCall{{ID: "c1", Type: "function", Function: ai.ToolCallFunction{Name: "read", Arguments: long}}}}}) + text := messages[0].Content[len(messages[0].Content)-1].Text + if !utf8.ValidString(text) { + t.Fatalf("clipped history is not valid UTF-8: %q", text) + } + if clipped := strings.TrimSuffix(text[strings.Index(text, " with ")+len(" with "):], "…"); len(clipped) > toolHistoryArgumentLimit { + t.Fatalf("clipped argument is %d bytes, over the %d-byte limit", len(clipped), toolHistoryArgumentLimit) + } +} diff --git a/internal/provider/toolless.go b/internal/provider/toolless.go new file mode 100644 index 0000000000..ad7466cf8e --- /dev/null +++ b/internal/provider/toolless.go @@ -0,0 +1,105 @@ +package provider + +import ( + "context" + "fmt" + "sync" + "time" +) + +// ── A MODEL THAT TAKES NO TOOLS IS NOT SENT ANY ───────────────────────────── +// +// The catalog publishes which request fields a model accepts, and a row that +// lists its parameters without "tools" has said the model takes no tool calls. +// Sending the belt anyway bought, on every turn of a 2026-09-28 conversation +// with microsoft/phi-4: a refused first attempt, a `Retry 1/1: removed tools` +// line nobody could read the reason in, and a request-size check that counted +// ~9k tokens of definitions the model was never going to receive — so the +// conversation was refused as too long at 15.6k when the request that would +// have gone out was 6k, and summarized to make room it already had. +// +// So the definitions are left off before the body is encoded, the size check +// measures what is really sent, and the person is told once, in words about +// what the model can do rather than about request fields. A row that lists no +// parameters has said nothing, and the tools go out as before: the refusal +// ladder is still there for a model the catalog does not know. + +// toollessMemo is which models this client has already said it is sending no +// tools to, so the notice is said once per model rather than on every turn — +// and which models the catalog could not describe but no provider would serve +// with tools, so the refusal is paid once per model rather than every turn. +// +// THE LEARNED HALF LAPSES AFTER A SERVING HOLD. It is not written to the +// quirks file: a provider that starts serving tools again gets another chance +// during a long-running client as well as after a restart. +type toollessMemo struct { + said sync.Map + refused sync.Map +} + +// learn records that the tools rung was what it took for this model. +func (m *toollessMemo) learn(model string) { + key := normalizeModel(model) + m.refused.Store(key, time.Now()) +} + +// learned says a provider has already refused this model's tools here. +func (m *toollessMemo) learned(model string) bool { + value, refused := m.refused.Load(normalizeModel(model)) + if !refused { + return false + } + at, dated := value.(time.Time) + if !dated || !at.After(time.Now().Add(-servingFactHold)) { + m.refused.CompareAndDelete(normalizeModel(model), value) + return false + } + return true +} + +// publishesNoTools says the catalog knows this model and says it takes no +// tool calls. Unknown is false. +func (c *Client) publishesNoTools(model string) bool { + if c.config.SupportsParameter == nil { + return false + } + supported, known := c.config.SupportsParameter(model, "tools") + return known && !supported +} + +// leaveOffTools marks this call's body to go without tool definitions when the +// model takes none, and tells the person the first time. +func (c *Client) leaveOffTools(ctx context.Context, model string, knobs callKnobs, carriesTools bool) callKnobs { + if !(c.publishesNoTools(model) || c.toolless.learned(model)) { + return knobs + } + if streamObserverFrom(ctx) != nil { + if _, said := c.toolless.said.LoadOrStore(normalizeModel(model), true); !said { + // NEWS ABOUT THE MODEL, NOT NARRATION ABOUT ONE REQUEST: it rides the + // row-news channel a surface keeps out of the folded work block, + // because a person who never opens that block still has to learn why + // the model will not touch their files. + Emit(ctx, StreamRowNews, toollessNotice(model)) + } + } + if carriesTools { + knobs.relaxed |= relaxTools + } + return knobs +} + +// toollessNotice is the one line a person reads about it. +func toollessNotice(model string) string { + return fmt.Sprintf("%s can't use tools, so it answers without them — it cannot read, search or change files", model) +} + +// encodeFailure is what a body that could not be built says. A refusal the +// size check made is already a sentence about the request and travels as it +// is; "marshal request:" in front of it read, on the person's screen, as +// though codeaf had failed to write JSON. +func encodeFailure(err error) error { + if _, refused := RefusalFrom(err); refused { + return err + } + return fmt.Errorf("marshal request: %w", err) +} diff --git a/internal/provider/toolless_history_contract_test.go b/internal/provider/toolless_history_contract_test.go new file mode 100644 index 0000000000..2e151803ae --- /dev/null +++ b/internal/provider/toolless_history_contract_test.go @@ -0,0 +1,90 @@ +package provider + +import ( + "context" + "encoding/json" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestToollessRequestCarriesToolHistoryAsReadableText(t *testing.T) { + history := []ai.Message{ + {Role: "user", Content: []ai.ContentPart{{Type: "text", Text: "read the file"}}}, + {Role: "assistant", Content: []ai.ContentPart{{Type: "text", Text: "I will look."}}, ToolCalls: []ai.ToolCall{{ID: "call_1", Type: "function", Function: ai.ToolCallFunction{Name: "read", Arguments: `{"path":"notes.txt"}`}}}}, + {Role: "tool", ToolCallID: "call_1", Content: []ai.ContentPart{{Type: "text", Text: "The answer is 42."}}}, + {Role: "user", Content: []ai.ContentPart{{Type: "text", Text: "What did it say?"}}}, + } + tools := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + client, recorded := newTestClient(t, Config{Model: "review/chat", SupportsParameter: func(model, parameter string) (bool, bool) { + return model != "review/chat" || parameter != "tools", true + }}) + if _, err := client.CompleteWithMessages(context.Background(), history, tools); err != nil { + t.Fatal(err) + } + if len(recorded.raw) != 1 { + t.Fatalf("requests = %d, want one", len(recorded.raw)) + } + got := string(recorded.raw[0]) + for _, absent := range []string{`"tool_calls"`, `"role":"tool"`, `"tool_call_id"`} { + if strings.Contains(got, absent) { + t.Fatalf("tool-less wire still carries %s: %s", absent, got) + } + } + for _, want := range []string{"I will look.", "read", "notes.txt", "The answer is 42."} { + if !strings.Contains(got, want) { + t.Fatalf("tool-less wire lost %q: %s", want, got) + } + } + client.config.Model = "review/tools" + before, err := json.Marshal(sanitizeMessages(history)) + if err != nil { + t.Fatal(err) + } + if _, err := client.CompleteWithMessages(context.Background(), history, tools); err != nil { + t.Fatal(err) + } + var body struct { + Messages json.RawMessage `json:"messages"` + } + if err := json.Unmarshal(recorded.raw[1], &body); err != nil { + t.Fatal(err) + } + if got := string(body.Messages); got != string(before) { + t.Fatalf("tools-model history bytes changed:\n%s\nwant:\n%s", got, before) + } +} + +func TestLearnedAndRelaxedToollessRequestsFlattenEarlierCalls(t *testing.T) { + for _, mode := range []string{"learned", "relaxed"} { + t.Run(mode, func(t *testing.T) { + client, err := NewClient(Config{BaseURL: "http://history.test", Model: "review/unknown", Direct: true}) + if err != nil { + t.Fatal(err) + } + knobs := callKnobs{} + if mode == "learned" { + client.toolless.learn("review/unknown") + } else { + knobs.relaxed = relaxTools + } + arguments := strings.Repeat("x", toolHistoryArgumentLimit+200) + request := &ai.Request{Model: "review/unknown", Tools: []ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}, Messages: []ai.Message{ + {Role: "assistant", ToolCalls: []ai.ToolCall{{ID: "c1", Function: ai.ToolCallFunction{Name: "read", Arguments: arguments}}}}, + {Role: "tool", ToolCallID: "c1", Content: []ai.ContentPart{{Type: "text", Text: "complete result"}}}, + }} + if mode == "learned" { + request.Tools = nil + } + body, err := client.encodeRequest(request, knobs) + if err != nil { + t.Fatal(err) + } + got := string(body) + if strings.Contains(got, `"tool_calls"`) || strings.Contains(got, `"role":"tool"`) || strings.Contains(got, arguments) || !strings.Contains(got, "complete result") { + t.Fatalf("%s history was not rendered or clipped: %s", mode, got) + } + }) + } +} diff --git a/internal/provider/toolless_hold_contract_test.go b/internal/provider/toolless_hold_contract_test.go new file mode 100644 index 0000000000..eae9a9d5c9 --- /dev/null +++ b/internal/provider/toolless_hold_contract_test.go @@ -0,0 +1,51 @@ +package provider + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestToollessMarkExpiresAndFailedToolsOffRetryDoesNotLearn(t *testing.T) { + var calls atomic.Int32 + var firstDone atomic.Bool + var secondHadTools atomic.Bool + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body map[string]any + _ = json.NewDecoder(r.Body).Decode(&body) + _, tools := body["tools"] + calls.Add(1) + if firstDone.Load() { + secondHadTools.Store(tools) + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"ok"},"finish_reason":"stop"}]}`)) + } else if tools { + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"error":{"message":"does not support tools"}}`)) + } else { + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write([]byte(`{"error":{"message":"temporary request failure"}}`)) + } + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: "review/unknown", Direct: true}) + if err != nil { + t.Fatal(err) + } + tool := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + _, _ = client.CompleteWithMessages(context.Background(), userMessages("first"), tool) + firstDone.Store(true) + _, err = client.CompleteWithMessages(context.Background(), userMessages("second"), tool) + if err != nil || !secondHadTools.Load() { + t.Fatalf("second request: error=%v tools=%v calls=%d", err, secondHadTools.Load(), calls.Load()) + } + client.toolless.refused.Store(normalizeModel("review/unknown"), time.Now().Add(-time.Hour)) + if client.toolless.learned("review/unknown") { + t.Fatal("expired tool-less mark remained active") + } +} diff --git a/internal/provider/toolless_notice_contract_test.go b/internal/provider/toolless_notice_contract_test.go new file mode 100644 index 0000000000..e0873c1f9f --- /dev/null +++ b/internal/provider/toolless_notice_contract_test.go @@ -0,0 +1,76 @@ +package provider + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sync" + "sync/atomic" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestNoToolsModelAnnouncesOnceWithEmptyAndFullBelt(t *testing.T) { + for _, withTools := range []bool{false, true} { + t.Run(map[bool]string{false: "empty", true: "full"}[withTools], func(t *testing.T) { + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + calls.Add(1) + var body map[string]any + _ = json.NewDecoder(r.Body).Decode(&body) + if _, has := body["tools"]; has { + t.Error("tool-less model received tools") + } + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"ok"},"finish_reason":"stop"}]}`)) + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: "review/no-tools", Direct: true, + SupportsParameter: func(_, p string) (bool, bool) { return p != "tools", true }}) + if err != nil { + t.Fatal(err) + } + var mu sync.Mutex + var news []string + ctx := WithStreamObserver(context.Background(), func(event StreamEvent) { + if event.Kind == StreamRowNews { + mu.Lock() + news = append(news, event.Delta) + mu.Unlock() + } + }) + var options []ai.Option + if withTools { + options = append(options, ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}})) + } + for i := 0; i < 2; i++ { + if _, err := client.CompleteWithMessages(ctx, userMessages("hello"), options...); err != nil { + t.Fatal(err) + } + } + mu.Lock() + defer mu.Unlock() + if len(news) != 1 || news[0] != toollessNotice("review/no-tools") || calls.Load() != 2 { + t.Fatalf("news=%q calls=%d, want one line and two calls", news, calls.Load()) + } + }) + } +} + +func TestLearnedToollessModelAnnouncesWithEmptyBelt(t *testing.T) { + var news []string + client, _ := NewClient(Config{BaseURL: "http://fixture.test", Model: "review/learned", Direct: true}) + client.toolless.learn("review/learned") + ctx := WithStreamObserver(context.Background(), func(event StreamEvent) { + if event.Kind == StreamRowNews { + news = append(news, event.Delta) + } + }) + for i := 0; i < 2; i++ { + client.leaveOffTools(ctx, "review/learned", callKnobs{}, false) + } + if len(news) != 1 || news[0] != toollessNotice("review/learned") { + t.Fatalf("news=%q, want one learned-model notice", news) + } +} diff --git a/internal/provider/toolless_notice_r3_contract_test.go b/internal/provider/toolless_notice_r3_contract_test.go new file mode 100644 index 0000000000..d74bfc2dc5 --- /dev/null +++ b/internal/provider/toolless_notice_r3_contract_test.go @@ -0,0 +1,37 @@ +package provider + +import ( + "context" + "net/http" + "net/http/httptest" + "testing" +) + +func TestUnobservedCallLeavesToollessNoticeForVisibleTurn(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + _, _ = w.Write([]byte(`{"choices":[{"message":{"role":"assistant","content":"ok"},"finish_reason":"stop"}]}`)) + })) + defer server.Close() + client, err := NewClient(Config{APIKey: "fixture", BaseURL: server.URL, Model: "review/no-tools", Direct: true, + SupportsParameter: func(_, p string) (bool, bool) { return p != "tools", true }}) + if err != nil { + t.Fatal(err) + } + if _, err := client.CompleteWithMessages(context.Background(), userMessages("helper")); err != nil { + t.Fatal(err) + } + var news []string + ctx := WithStreamObserver(context.Background(), func(e StreamEvent) { + if e.Kind == StreamRowNews { + news = append(news, e.Delta) + } + }) + for i := 0; i < 2; i++ { + if _, err := client.CompleteWithMessages(ctx, userMessages("visible chat")); err != nil { + t.Fatal(err) + } + } + if len(news) != 1 || news[0] != toollessNotice("review/no-tools") { + t.Fatalf("visible notices = %q, want one", news) + } +} diff --git a/internal/provider/toolless_test.go b/internal/provider/toolless_test.go new file mode 100644 index 0000000000..e9e472473a --- /dev/null +++ b/internal/provider/toolless_test.go @@ -0,0 +1,151 @@ +package provider + +import ( + "context" + "strings" + "sync" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +// A MODEL THE CATALOG SAYS TAKES NO TOOLS IS SENT NONE: no refused first +// attempt, no retry line, one plain notice for the model, and a size check +// that does not count definitions that never go out (2026-09-28, +// microsoft/phi-4: refused as 15.6k tokens when 6k would have been sent). +func TestAModelThatTakesNoToolsIsSentNone(t *testing.T) { + recorded := &capture{} + config := attributedConfig(streamedRefusal(func(body map[string]any) bool { + _, hasTools := body["tools"] + return !hasTools + }, recorded)) + config.SupportsParameter = func(_, parameter string) (bool, bool) { return parameter != "tools", true } + client, err := NewClient(config) + if err != nil { + t.Fatal(err) + } + var mu sync.Mutex + var notices []string + ctx := WithStreamObserver(context.Background(), func(event StreamEvent) { + if event.Kind == StreamNotice || event.Kind == StreamRowNews { + mu.Lock() + notices = append(notices, event.Delta) + mu.Unlock() + } + }) + // A schema big enough that, counted, it alone would overflow the window. + tools := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{ + Name: "read", Parameters: map[string]any{"description": strings.Repeat("schema ", 6000)}, + }}}) + ctx = WithContextBudget(ctx, ContextBudget{Window: 8192, Reserve: 512}) + for turn := 1; turn <= 2; turn++ { + response, err := client.CompleteWithMessages(ctx, userMessages("hello"), tools) + if err != nil { + t.Fatalf("turn %d: %v", turn, err) + } + if response.Text() != "ok" { + t.Fatalf("turn %d answered %q", turn, response.Text()) + } + } + if len(recorded.bodies) != 2 { + t.Fatalf("requests = %d, want one per turn and no refused attempt", len(recorded.bodies)) + } + mu.Lock() + defer mu.Unlock() + if len(notices) != 1 || notices[0] != toollessNotice("sim/model") { + t.Fatalf("notices = %#v, want the one tool-less line once", notices) + } +} + +// A MODEL THE CATALOG DOES NOT KNOW still carries its tools, and the ladder +// says why it took them off when an endpoint refuses them. +func TestAnUnknownModelStillCarriesItsTools(t *testing.T) { + recorded := &capture{} + config := attributedConfig(streamedRefusal(func(map[string]any) bool { return true }, recorded)) + config.SupportsParameter = func(string, string) (bool, bool) { return false, false } + client, err := NewClient(config) + if err != nil { + t.Fatal(err) + } + tools := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + if _, err := client.CompleteWithMessages(context.Background(), userMessages("hello"), tools); err != nil { + t.Fatal(err) + } + if _, hasTools := recorded.body(0)["tools"]; !hasTools { + t.Fatal("an unknown model was sent no tools") + } +} + +// A size refusal is a sentence about the request, not a failure to write JSON. +func TestASizeRefusalIsNotCalledAMarshalFailure(t *testing.T) { + client, _ := NewClient(Config{APIKey: "test", BaseURL: "http://budget.test", Model: "budget/words", Direct: true}) + ctx := WithContextBudget(context.Background(), ContextBudget{Window: 8192, Reserve: 2048}) + _, err := client.CompleteWithMessages(ctx, userMessages(strings.Repeat("input ", 9000))) + if err == nil || strings.Contains(err.Error(), "marshal request") { + t.Fatalf("error = %v", err) + } +} + +// A REFUSAL OF TOOLS IS PAID ONCE PER RUN for a model the catalog does not +// know: the first turn is refused and retried without them, and every later +// turn goes without them from the start, with no second line said about it. +func TestAToolsRefusalIsRememberedForTheRun(t *testing.T) { + recorded := &capture{} + config := attributedConfig(streamedRefusal(func(body map[string]any) bool { + _, hasTools := body["tools"] + return !hasTools + }, recorded)) + config.SupportsParameter = func(string, string) (bool, bool) { return false, false } + client, err := NewClient(config) + if err != nil { + t.Fatal(err) + } + var mu sync.Mutex + var said []string + ctx := WithStreamObserver(context.Background(), func(event StreamEvent) { + if event.Kind == StreamNotice || event.Kind == StreamRowNews { + mu.Lock() + said = append(said, event.Delta) + mu.Unlock() + } + }) + tools := ai.WithTools([]ai.ToolDefinition{{Type: "function", Function: ai.ToolFunction{Name: "read"}}}) + if _, err := client.CompleteWithMessages(ctx, userMessages("first"), tools); err != nil { + t.Fatal(err) + } + first := len(recorded.bodies) + mu.Lock() + lines := len(said) + mu.Unlock() + if first < 2 || lines == 0 { + t.Fatalf("turn one: %d requests and %d lines, want a refusal, a retry and its line", first, lines) + } + if _, err := client.CompleteWithMessages(ctx, userMessages("second"), tools); err != nil { + t.Fatal(err) + } + if got := len(recorded.bodies) - first; got != 1 { + t.Fatalf("turn two made %d requests, want one without tools", got) + } + if _, hasTools := recorded.body(len(recorded.bodies) - 1)["tools"]; hasTools { + t.Fatal("turn two carried the tools again") + } + mu.Lock() + defer mu.Unlock() + if len(said) != lines { + t.Fatalf("turn two said %q, want nothing new", said[lines:]) + } +} + +// A body with no tools carries no tool_choice either. +func TestNoToolsMeansNoToolChoice(t *testing.T) { + client, _ := NewClient(Config{BaseURL: "http://choice.test", Model: "choice/model", Direct: true}) + request := &ai.Request{Model: "choice/model", Messages: userMessages("hello")} + _ = ai.WithTools(nil)(request) + body, err := client.encodeRequest(request, callKnobs{}) + if err != nil { + t.Fatal(err) + } + if strings.Contains(string(body), "tool_choice") { + t.Fatalf("body = %s", body) + } +} diff --git a/internal/provider/wire.go b/internal/provider/wire.go index cfc6de005e..fd670a53de 100644 --- a/internal/provider/wire.go +++ b/internal/provider/wire.go @@ -328,6 +328,13 @@ func (c *Client) encodeRequest(request *ai.Request, knobs callKnobs) ([]byte, er scrubbed.Tools = nil scrubbed.ToolChoice = nil } + // A CHOICE AMONG NO TOOLS IS NOT SENT. The SDK sets "auto" beside even an + // empty belt — a conversation on a model with no tools carries one + // (internal/session's chatpage.go) — and an endpoint that validates the + // pair refuses tool_choice without tools. + if len(scrubbed.Tools) == 0 { + scrubbed.ToolChoice = nil + } if knobs.relaxed.has(relaxResponseFormat) { scrubbed.ResponseFormat = nil } @@ -343,6 +350,12 @@ func (c *Client) encodeRequest(request *ai.Request, knobs callKnobs) ([]byte, er } model := c.modelFor(&scrubbed) + // A request without definitions must not replay tool protocol messages to + // an endpoint that cannot accept them. The conversion precedes encoding so + // the budget counts the text that actually goes out. + if len(scrubbed.Tools) == 0 && (knobs.relaxed.has(relaxTools) || c.publishesNoTools(model) || c.toolless.learned(model)) { + scrubbed.Messages = readableToolHistory(scrubbed.Messages) + } // The dialect is resolved once per encode rather than cached on the client, // because the model can be pinned per request by the router and the learned @@ -372,10 +385,12 @@ func (c *Client) encodeRequest(request *ai.Request, knobs callKnobs) ([]byte, er Tools: tools, PromptCacheKey: knobs.cacheKey, } + // The knob is decided here and written after the context budget below, + // which may shrink the thinking budget to what the window leaves. + sentEffort, thinking := EffortNone, 0 if !knobs.relaxed.has(relaxReasoning) { - wire.Reasoning = reasoningFor( - c.resolveEffort(model, knobs.effort), - c.resolveReasoningBudget(model, knobs.effort)) + sentEffort = c.resolveEffort(model, knobs.effort) + thinking = c.resolveReasoningBudget(model, knobs.effort) } // The ceiling that travels is the caller's answer plus the thinking pass's // room (thinking.go's ceilingFor), read here and again by the transport so @@ -399,6 +414,13 @@ func (c *Client) encodeRequest(request *ai.Request, knobs callKnobs) ([]byte, er // refusal carries none — so this is set at the one line that puts the // object on the bytes, true or false, and nowhere earlier (prefcarry.go). c.prefWentOut(wire.Provider != nil) + ceiling, hasCeiling, thinking, err = c.budgetWire(&scrubbed, knobs, messages, tools, wire.Provider, ceiling, hasCeiling, thinking) + if err != nil { + return nil, err + } + if !knobs.relaxed.has(relaxReasoning) { + wire.Reasoning = reasoningFor(sentEffort, thinking) + } if hasCeiling { if needsMaxCompletionTokens(model) && isVouchedRewriteEndpoint(c.config.BaseURL) { wire.MaxCompletionTokens = &ceiling diff --git a/internal/remote/callclass.go b/internal/remote/callclass.go index c48380e0f8..b095d1d31c 100644 --- a/internal/remote/callclass.go +++ b/internal/remote/callclass.go @@ -27,7 +27,17 @@ package remote // the classes says why the longer window it was given first had to come back // out. // -// AND LETTING EITHER OF THEM OVERTAKE AN ORDERED CALL COSTS NOTHING, which is +// WORK is a long piece of the engine's own labour that a surface asked for and +// nothing else waits on: [MethodCompact], which may ask the model for a summary +// and take a minute doing it. It ran on the ordered lane until 2026-09-28, and +// everything the person sent while it ran queued behind it and waited out its +// own deadline there. It owes no order for the reason below: the surface asks +// for it from a command goroutine of its own, so nothing it sends was ever +// ordered after it, and the conversation guards itself — a summary is spliced +// in only if the region it read is unchanged (internal/session's +// compact_summary.go), which is what an in-process surface already relied on. +// +// AND LETTING ANY OF THEM OVERTAKE AN ORDERED CALL COSTS NOTHING, which is // the whole licence for this split and is worth stating plainly: EVERY CALL ON // THIS WIRE IS A SYNCHRONOUS ROUND TRIP ([Client.callAnswered] waits for the // result frame). A caller that makes two calls therefore has the first one's @@ -72,6 +82,7 @@ const ( classOrdered callClass = iota classGetter classAct + classWork ) // road is WHERE a class of call runs, and this is the one place a class is @@ -87,8 +98,8 @@ const ( // inOrder is the connection's one ordered lane: off the reader, and in the // order the frames arrived (orderedlane.go). inOrder road = iota - // onItsOwn is a goroutine per call. A getter and a small act owe nothing to - // each other, so neither owes a queue. + // onItsOwn is a goroutine per call. A getter, a small act and a piece of + // work owe nothing to each other, so none of them owes a queue. onItsOwn ) @@ -160,6 +171,8 @@ func classify(method string) callClass { MethodPlanNote, MethodPlanPause, MethodPlanResume, MethodPlanCancel, MethodPlanAmend, MethodPlanPriority, MethodTyping: return classAct + case MethodCompact: + return classWork default: return classOrdered } diff --git a/internal/remote/client.go b/internal/remote/client.go index 4b04d10b6a..de8de12b25 100644 --- a/internal/remote/client.go +++ b/internal/remote/client.go @@ -1146,9 +1146,20 @@ func (c *Client) late() error { if roaming { return errors.New(c.roamingRefusal()) } - return errors.New(c.where() + lateCallTail) + return lateError{where: c.where()} } +// ErrLate is what a call that outlived this end's patience matches with +// [errors.Is]. The engine may still be doing the work, so a surface that can +// say "still running" rather than "failed" reads it through this. +var ErrLate = errors.New("the engine did not answer in time") + +// lateError is [Client.late]'s sentence, and it is [ErrLate]. +type lateError struct{ where string } + +func (e lateError) Error() string { return e.where + lateCallTail } +func (e lateError) Is(target error) bool { return target == ErrLate } + // forget drops a call nobody is waiting for any more. func (c *Client) forget(id uint64) { c.mu.Lock() @@ -1772,8 +1783,22 @@ func (a *Agent) StopWork() error { } // Compact runs a compaction pass on the far side. +// +// IT WAITS AS LONG AS A PASS CAN TAKE, not [callDeadline]. A pass may ask the +// model for a summary, which on a slow model is longer than ten seconds, and a +// surface that gave up sooner said "did not answer in time" about a pass that +// landed a moment later. The surface asks from a command rather than from its +// update loop, so the longer wait is a line saying "compacting…", never a +// terminal that stops drawing. It runs beside the ordered lane (callclass.go's +// [classWork]), so nothing the person sends meanwhile queues behind it. func (a *Agent) Compact(ctx context.Context) error { - _, err := a.c.call(ctx, MethodCompact, nil) + result, err := a.c.callWithin(ctx, MethodCompact, nil, session.CompactPatience) + if err == nil && len(result) > 0 && string(result) != "null" { + var why string + if json.Unmarshal(result, &why) == nil && why != "" { + return &session.SummarySkipped{Why: why} + } + } return err } diff --git a/internal/remote/compact_test.go b/internal/remote/compact_test.go new file mode 100644 index 0000000000..3c5222c235 --- /dev/null +++ b/internal/remote/compact_test.go @@ -0,0 +1,63 @@ +package remote + +import ( + "context" + "testing" + + "github.com/Agent-Field/codeaf/internal/session" +) + +type compactingAgent struct{ *fakeAgent } + +func (a compactingAgent) Compact(context.Context) error { + a.mu.Lock() + defer a.mu.Unlock() + a.tokens = 2000 + return nil +} + +func TestCompactPublishesReducedSizeBeforeReply(t *testing.T) { + agent := &fakeAgent{model: "test/model", tokens: 30000} + engine := engineOn(agent) + engine.Agent = compactingAgent{agent} + link := dialAgent(t, engine) + link.hello(Hello{Version: Version}) + link.ok(1, MethodCompact, nil) + if len(link.stated) == 0 { + t.Fatal("compact replied without publishing fresh facts") + } + facts := decode[FactsPush](t, link.stated[len(link.stated)-1].Payload) + if facts.Facts.ContextTokens != 2000 { + t.Fatalf("compact replied with stale size %d", facts.Facts.ContextTokens) + } +} + +type compactingWithoutSummary struct{ *fakeAgent } + +func (a compactingWithoutSummary) Compact(context.Context) error { + a.mu.Lock() + a.tokens = 2000 + a.mu.Unlock() + return &session.SummarySkipped{Why: "provider unavailable"} +} + +func TestCompactServerCarriesSkippedSummaryInSuccessfulResult(t *testing.T) { + agent := &fakeAgent{model: "test/model", tokens: 30000} + engine := engineOn(agent) + engine.Agent = compactingWithoutSummary{agent} + link := dialAgent(t, engine) + link.hello(Hello{Version: Version}) + result := link.ok(1, MethodCompact, nil) + if why := decode[string](t, result.Payload); why != "provider unavailable" { + t.Fatalf("optional success result = %q", why) + } +} + +func TestCompactClientReadsSkippedSummaryResult(t *testing.T) { + client, engine := newEngine(t) + engine.answers[MethodCompact] = "provider unavailable" + why, ok := session.SummarySkippedWhy(client.Agent().Compact(context.Background())) + if !ok || why != "provider unavailable" { + t.Fatalf("remote compact result = %q, skipped=%v", why, ok) + } +} diff --git a/internal/remote/orderedlane.go b/internal/remote/orderedlane.go index 5591d00a13..429f837df8 100644 --- a/internal/remote/orderedlane.go +++ b/internal/remote/orderedlane.go @@ -28,9 +28,10 @@ import "sync" // a surface whose call reaches [callDeadline] is free to send another, so the // true bound is its outstanding calls plus one per deadline per goroutine — which // is what "bounded by that surface's own patience" means; and after a long -// [MethodCompact] the lane replays ordered calls nobody is waiting for any more, +// ordered call the lane replays ordered calls nobody is waiting for any more, // which is what the socket buffer did before this type existed and is therefore -// the same behaviour rather than a new one. +// the same behaviour rather than a new one. [MethodCompact], the longest there +// was, no longer rides here (callclass.go's [classWork]). // // It is a type rather than a channel and a goroutine written inline because // "run these in the order they came, off the goroutine that received them" is a diff --git a/internal/remote/road_test.go b/internal/remote/road_test.go index c825f4ac7a..52af6e9b08 100644 --- a/internal/remote/road_test.go +++ b/internal/remote/road_test.go @@ -1,6 +1,8 @@ package remote import ( + "context" + "errors" "strings" "sync" "testing" @@ -127,6 +129,9 @@ func TestADeadlineOnALiveConnectionDoesNotSayTheConnectionIsGone(t *testing.T) { if !strings.Contains(err.Error(), strings.TrimSpace(lateCallTail)) { t.Fatalf("the late call did not say so: %v", err) } + if !errors.Is(err, ErrLate) { + t.Fatalf("a surface cannot tell the late call from a failure: %v", err) + } if client.Err() != nil { t.Fatalf("the connection was buried: %v", client.Err()) } @@ -149,11 +154,16 @@ func TestTheCallClassKeepsAKeystrokeOffTheReader(t *testing.T) { // between Submit and the first request leaving used to run there and // everything behind it on this socket waited (callclass.go); what an // ordered call owes is an order, and the lane is what gives it. - for _, method := range []string{MethodSubmit, MethodCompact, MethodSetModel} { + for _, method := range []string{MethodSubmit, MethodSetModel} { if classify(method).road() != inOrder { t.Fatalf("%s owes an order and is not on the ordered lane", method) } } + // A COMPACTION IS WORK THAT OWES NOBODY AN ORDER. It may ask the model for + // a summary, and a send queued behind it waited out its own deadline. + if classify(MethodCompact) != classWork || classify(MethodCompact).road() != onItsOwn { + t.Fatal("a compaction is back on the ordered lane, where a send waits behind its summary") + } // AND AN ACT IS NOT A CLASS OF ITS OWN ON THE CLOCK, which is a law with a // measurement behind it (callclass.go): a longer window for a keystroke buys // a terminal that stops drawing, because the question block asks its door @@ -162,3 +172,77 @@ func TestTheCallClassKeepsAKeystrokeOffTheReader(t *testing.T) { t.Fatal("the two off-reader classes are no longer told apart") } } + +// heldCompaction is a fake engine whose Compact does not return until the test +// lets it, the shape of a pass waiting on a slow model's summary. +type heldCompaction struct { + *fakeAgent + entered chan struct{} + release chan struct{} + once sync.Once +} + +func (h *heldCompaction) Compact(context.Context) error { + h.once.Do(func() { close(h.entered) }) + <-h.release + return nil +} + +// TestASendIsNotHeldBehindACompactionWaitingOnItsSummary is the measured +// defect of 2026-09-28: /compact asked deepseek-v3.2 for a summary that took +// twenty seconds, and the message the person sent meanwhile sat on the ordered +// lane behind it, past its own deadline. The compaction is held here; the send +// must open its turn anyway, and the compaction must still land afterwards. +func TestASendIsNotHeldBehindACompactionWaitingOnItsSummary(t *testing.T) { + far := &heldCompaction{ + fakeAgent: &fakeAgent{model: "m"}, + entered: make(chan struct{}), + release: make(chan struct{}), + } + loop, err := Loopback(Hello{Version: Version}, Options{Boot: func(Hello) (*Engine, error) { + return &Engine{Agent: far, Workspace: "/srv/app", SessionFile: "/srv/app/j.jsonl"}, nil + }}) + if err != nil { + t.Fatalf("dial the loopback: %v", err) + } + t.Cleanup(func() { + select { + case <-far.release: + default: + close(far.release) + } + _ = loop.Close() + }) + + compacted := make(chan error, 1) + go func() { compacted <- loop.Client.Agent().Compact(context.Background()) }() + select { + case <-far.entered: + case <-time.After(time.Second): + t.Fatal("the compaction never reached the engine") + } + + sent := make(chan error, 1) + go func() { + _, err := loop.Client.Agent().Submit(context.Background(), "you there?") + sent <- err + }() + select { + case err := <-sent: + if err != nil { + t.Fatalf("the send was refused: %v", err) + } + case <-time.After(2 * time.Second): + t.Fatal("the send did not return; it is waiting behind the compaction") + } + + close(far.release) + select { + case err := <-compacted: + if err != nil { + t.Fatalf("the compaction failed once released: %v", err) + } + case <-time.After(2 * time.Second): + t.Fatal("the compaction never answered after it was released") + } +} diff --git a/internal/remote/server.go b/internal/remote/server.go index ed946b484d..b5b66e3aca 100644 --- a/internal/remote/server.go +++ b/internal/remote/server.go @@ -2571,7 +2571,17 @@ func (s *server) invoke(call Frame) (out json.RawMessage, err error) { return mustJSON(door.AnswerLaneOffer(yes)), nil case MethodCompact: - return nil, agent.Compact(context.Background()) + err := agent.Compact(context.Background()) + if why, skipped := session.SummarySkippedWhy(err); skipped { + // An older surface ignores a successful call's optional result and + // still reads a true success, while a newer one can show the reason. + sess.announce() + return mustJSON(why), nil + } + // The command's reply follows the new size, so hosted surfaces show + // the same before/after reading as a local conversation. + sess.announce() + return nil, err case MethodClose: // The surface said goodbye politely, and it is saying it about the diff --git a/internal/remote/summarized_wire_test.go b/internal/remote/summarized_wire_test.go new file mode 100644 index 0000000000..d46b769c21 --- /dev/null +++ b/internal/remote/summarized_wire_test.go @@ -0,0 +1,38 @@ +package remote + +import ( + "encoding/json" + "strings" + "testing" + + "github.com/Agent-Field/codeaf/internal/session" +) + +// A PASS'S SUMMARY COUNT CROSSES THE LINK. A surface over --host decides +// whether a compaction's line stands by [session.Event.Summarized], so a wire +// that dropped it would fold a summary away there while the same pass stood on +// the machine that ran it. And a pass with no summary sends no field at all, +// which is the bytes an older engine sends for every pass. +func TestACompactionsSummaryCountCrossesTheWire(t *testing.T) { + payload, err := json.Marshal(WireEvent(session.Event{ + Kind: session.EventCompacted, Hint: "compacted · summarized 4 messages", Summarized: 4, + })) + if err != nil { + t.Fatalf("marshal: %v", err) + } + var wire EventWire + if err := json.Unmarshal(payload, &wire); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if got := wire.Unwire().Summarized; got != 4 { + t.Fatalf("the summary count did not survive the link: got %d, want 4", got) + } + + free, err := json.Marshal(WireEvent(session.Event{Kind: session.EventCompacted, Hint: "compacted · folded 3 messages"})) + if err != nil { + t.Fatalf("marshal: %v", err) + } + if strings.Contains(string(free), "Summarized") { + t.Fatalf("a free pass put the field on the wire: %s", free) + } +} diff --git a/internal/remote/surfacedoors_law_test.go b/internal/remote/surfacedoors_law_test.go index e95fb1998b..8f2f7edd92 100644 --- a/internal/remote/surfacedoors_law_test.go +++ b/internal/remote/surfacedoors_law_test.go @@ -277,7 +277,7 @@ func TestEveryStreamOpenerIsOrdered(t *testing.T) { // AND NO CLASS AT ALL RUNS ON THE GOROUTINE THAT READS THE SOCKET. The road // type has no third value, so this is a statement about the two it has // rather than a check somebody could forget to extend. - for _, class := range []callClass{classOrdered, classGetter, classAct} { + for _, class := range []callClass{classOrdered, classGetter, classAct, classWork} { if class.road() != inOrder && class.road() != onItsOwn { t.Errorf("a class found a road that is not one of the two") } diff --git a/internal/session/agent.go b/internal/session/agent.go index 3fccee1d65..5346319247 100644 --- a/internal/session/agent.go +++ b/internal/session/agent.go @@ -12,6 +12,7 @@ import ( "github.com/Agent-Field/agentfield/sdk/go/ai" "github.com/Agent-Field/codeaf/internal/buildinfo" "github.com/Agent-Field/codeaf/internal/effort" + "github.com/Agent-Field/codeaf/internal/env" "github.com/Agent-Field/codeaf/internal/guard" lanes "github.com/Agent-Field/codeaf/internal/lane" "github.com/Agent-Field/codeaf/internal/modelsource" @@ -115,13 +116,17 @@ func newAgent(config Config, client Completer) (*Agent, error) { if config.newerBuild == nil { config.newerBuild = buildinfo.StaleNotice } - // WHICH OF THE TWO FIXED PREFIXES THIS SESSION SENDS, SETTLED ONCE AND - // BEFORE ANYTHING IS BUILT FROM IT (promptprofile.go). It is derived rather - // than configured — the model's window and the crew's worker seat are the - // two facts — and it is settled HERE, above the render, because the page, - // the belt, the shelf and the memory reflex are all built from this one - // config and a profile resolved twice is a profile that can answer twice. + // Settle the launch preference before building either the page or belt. + // Explicit pins remain fixed; automatic profiles follow the selected + // window at later request boundaries (promptprofile_live.go). config.profile = settlePromptProfile(config) + _, pinned := promptProfileWord(env.Get(promptProfileEnv)) + _, chosen := promptProfileWord(config.PromptProfile) + config.liveProfile = &livePromptProfile{auto: !pinned && !chosen} + if !config.liveProfile.auto { + config.liveProfile.launch = chosenPromptProfile(config) + } + config.liveProfile.current.Store(config.profile) system, own := config.System, false if strings.TrimSpace(system) == "" { system, own = renderSystem(config), true @@ -145,10 +150,8 @@ func newAgent(config Config, client Completer) (*Agent, error) { } agent.presentation = &presentationIndex{} agent.cacheKey = sessionCacheKey(agent.id) - // WHAT IS ALREADY KNOWN ABOUT THIS MODEL'S REAL WINDOW, before the first - // check. The memo survives processes (internal/provider's ServedWindow), so - // a model that refused an over-long prompt last week is capped from this - // session's first turn rather than from its first refusal (loop.go). + // Start with no session-local endpoint limit; the provider applies its + // durable evidence when it encodes a request. agent.noteModelWindow(agent.model) // AND WHO THIS SESSION IS WORKING FOR, before anything else is built // (principal.go). It is written once here and never again, which is what @@ -162,13 +165,10 @@ func newAgent(config Config, client Completer) (*Agent, error) { // store has to exist before the tools are assembled (memory.go). The // background lifetime is minted with it, because a pass started by the first // turn has to have somewhere to be cancelled from. - // THE PREDICATE IS [Config.hasStore] AND NOT THE FIELD, because the field is - // two things: the conversation's own record, which every shape writes and - // reads, and the writable memory this brain is, which a lean prefix does not - // have (promptprofile.go). The belt and the page are built from that same - // predicate a moment later, which is what stops them disagreeing about - // whether `remember` exists. - if config.hasStore() { + // Automatic lean sessions keep a dormant brain so a later switch back to + // full can enable memory without changing a pointer background readers + // hold. remembers gates every memory entry point by the live profile. + if config.Memory != nil && (config.hasStore() || config.liveProfile.auto || config.profile.chat()) { agent.memory = newMemoryBrain(config.Memory) agent.memoryCtx, agent.memoryStop = context.WithCancel(context.Background()) } @@ -620,14 +620,14 @@ func (a *Agent) Usage() Usage { // every tool result, the arguments of every call, all the bytes a surface // counting words cannot see. When the transcript has grown since (a 300KB file // read that has not been sent yet), the content estimate is larger and wins. See -// [Agent.estimateTokensLocked] for why it is the max of the two. +// [Agent.meterTokensLocked] for why it is the max of the two. // // Zero is a session that has neither sent nor recorded anything, which is the // only case where "nothing" is true. func (a *Agent) ContextTokens() int { a.mu.Lock() defer a.mu.Unlock() - return a.estimateTokensLocked() + return a.meterTokensLocked() } // Model returns the model the next request will use. @@ -2183,17 +2183,18 @@ func (a *Agent) Compact(ctx context.Context) error { // // It was one extra instruction for the summarizer — `/compact keep the API // decisions and the failing test`, a person saying which part of a lossy summary -// had to survive. There is no summarizer any more (loop.go): a pass stubs tool -// results and folds assistant work, and neither of those is a judgement anybody -// can steer. The kept content is the same whatever is typed after /compact — -// every user message, the recent tail, and the state card — so there is nothing -// for a focus to protect that is not already protected. +// had to survive. The summarizer was deleted on 2026-08-18 and came back on +// 2026-09-28 only as a pass's last rung (compact_summary.go), and the surface +// passes no focus to it: `/compact` takes no argument. // // The door stays open with its signature unchanged because the surface calls it // (internal/tui3), and a person who types the old form gets the pass they asked // for rather than an error about a machine that used to exist. func (a *Agent) CompactWithFocus(ctx context.Context, _ string) error { - _, err := a.compact(ctx, nil) + _, skipped, err := a.compactWithPolicyResult(ctx, nil, a.requestedCompactPolicy()) + if err == nil && skipped != "" { + return &SummarySkipped{Why: skipped} + } return err } @@ -4808,6 +4809,11 @@ func shapeEntries(messages []ai.Message, journal *sessionFile, indexes ...*prese continue } role := msg.Role + // A SUMMARY IS THE SESSION'S RECORD OF WHAT WENT, not something anybody + // typed, and it is drawn as the divider a "note" is (compact_summary.go). + if role == "user" && strings.HasPrefix(messageContentText(msg), summaryNotePrefix) { + role = "note" + } if role == "user" && journal.isNote(msg) { // A LINE THE SESSION WROTE IS NOT THE PERSON'S. It is user-role in the // transcript because that is the only role the model can be told diff --git a/internal/session/card.go b/internal/session/card.go index f1dfc8d7af..668c2dc05e 100644 --- a/internal/session/card.go +++ b/internal/session/card.go @@ -10,7 +10,8 @@ package session // exchange (memory.go), and it was already returning a state delta that nothing // read ([reflex.ExtractResult.State]). Folding that delta into a small file // amortizes the whole cost across the turns that produced it, and leaves -// compaction with nothing to do but rearrange. +// compaction to rearrange in all but the last resort, when a pass that cannot +// shrink the conversation any other way writes a summary (compact_summary.go). // // ── WHY THIS IS NOT state.json ── // diff --git a/internal/session/chatpage.go b/internal/session/chatpage.go new file mode 100644 index 0000000000..997ad16297 --- /dev/null +++ b/internal/session/chatpage.go @@ -0,0 +1,66 @@ +package session + +import ( + _ "embed" + "fmt" + "runtime" + "strings" + "time" +) + +// ── THE PAGE A MODEL WITH NO TOOLS READS ──────────────────────────────────── +// +// A model the catalog says takes no tool calls is sent no tools +// (internal/provider's toolless.go). The working page was still sent with it: +// on 2026-09-28 microsoft/phi-4 carried 23,640 characters of system prompt — +// tool policy, workflow, delegation, the facts about files and jobs — about a +// third of its 16k window, every word of it about hands it did not have. A +// capability that cannot work is absent, and so is the page about it. +// +// So a conversation on such a model reads this page instead: who it is, what it +// cannot do and what to say about that, how to answer, the codeaf messages it +// must not answer as the person's, and the project facts that are true of this +// minute. The belt is empty ([Agent.belt]) and memory is off, as on a lean +// prefix. It lasts while the model does: the conversation's next request on a +// model that can use tools gets the working page back (promptprofile_live.go). + +//go:embed prompts/chat.md +var chatPrompt string + +// chatPage is the whole system prompt of a tool-less conversation. +// +// THE MESSAGES SECTION IS CUT FROM THE WORKING PAGE rather than written twice, +// so the list of codeaf's own user-role tags cannot drift between the two +// (messagesfromcodeaf_test.go pins it on system.md). +func chatPage(config Config, now time.Time) string { + var out strings.Builder + out.WriteString(strings.TrimRight(chatPrompt, "\n")) + if messages := pageSection(systemPrompt, "Messages from codeaf"); messages != "" { + out.WriteString("\n\n") + out.WriteString(messages) + } + out.WriteString("\n\n# Project\n") + fmt.Fprintf(&out, "- Workstation: %s/%s\n", runtime.GOOS, runtime.GOARCH) + fmt.Fprintf(&out, "- Working directory: %s\n", config.Workspace) + out.WriteString(nowLine(now)) + return out.String() +} + +// pageSection is one `# ` section of a page, heading included, and "" when the +// page has no such heading. +func pageSection(page, heading string) string { + var kept []string + inside := false + for _, line := range strings.Split(page, "\n") { + if strings.HasPrefix(line, "# ") { + if inside { + break + } + inside = strings.TrimSpace(strings.TrimPrefix(line, "# ")) == heading + } + if inside { + kept = append(kept, line) + } + } + return strings.TrimRight(strings.Join(kept, "\n"), "\n") +} diff --git a/internal/session/chatpage_test.go b/internal/session/chatpage_test.go new file mode 100644 index 0000000000..b063e7091a --- /dev/null +++ b/internal/session/chatpage_test.go @@ -0,0 +1,128 @@ +package session + +import ( + "context" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +// toolsFor is a catalog that says "chat/model" takes no tools and every other +// model does. +func toolsFor(model, parameter string) (bool, bool) { + if parameter != "tools" { + return true, true + } + return model != "chat/model", true +} + +// A MODEL WITH NO TOOLS READS A PAGE WITHOUT THEM: no tool policy, no belt, no +// memory, and a plain sentence about what it cannot do. +func TestAModelWithNoToolsReadsTheChatPage(t *testing.T) { + t.Setenv(promptProfileEnv, "") + agent, _ := newTestAgent(t, &scriptedCompleter{}, func(c *Config) { + buildShippedConversation(t, c) + c.System = "" + c.Model = "chat/model" + c.SupportsParameter = toolsFor + }) + if !agent.config.promptProfile().chat() { + t.Fatal("a model with no tools did not get the chat profile") + } + page := systemOf(agent) + for _, want := range []string{"This model has no tools in codeaf", "# Messages from codeaf", "# Project"} { + if !strings.Contains(page, want) { + t.Fatalf("chat page is missing %q", want) + } + } + for _, gone := range []string{"# Tool Policy", "# Putting more hands on the work", "# Session facts"} { + if strings.Contains(page, gone) { + t.Fatalf("chat page still carries %q", gone) + } + } + if len(agent.beltDefinitions()) != 0 || agent.remembers() { + t.Fatalf("belt %d tools, memory %v: want none and off", len(agent.beltDefinitions()), agent.remembers()) + } + if full := len(renderSystem(Config{Workspace: agent.config.Workspace, Model: "tool/model"})); len(page)*3 > full { + t.Fatalf("chat page is %d bytes against a working page of %d", len(page), full) + } +} + +// THE PAGE FOLLOWS THE MODEL: onto a model with no tools and back again, with +// an explicitly loaded capability returning with the tools. +func TestSwitchingToAModelWithNoToolsSwapsThePageAndBack(t *testing.T) { + t.Setenv(promptProfileEnv, "") + agent := leanShapedAgent(t) + agent.config.SupportsParameter = toolsFor + if _, failed := agent.loadCapability("tasks"); failed { + t.Fatal("load tasks") + } + if err := agent.preparePromptProfile("chat/model"); err != nil { + t.Fatal(err) + } + if !agent.config.promptProfile().chat() || len(agent.beltDefinitions()) != 0 { + t.Fatalf("after moving to a model with no tools: profile %q, %d tools", agent.config.promptProfile(), len(agent.beltDefinitions())) + } + if !strings.Contains(systemOf(agent), "This model has no tools in codeaf") { + t.Fatal("the page did not change with the model") + } + if err := agent.preparePromptProfile(agent.Model()); err != nil { + t.Fatal(err) + } + if agent.config.promptProfile().chat() || !agent.hasTool("propose_task") { + t.Fatal("coming back did not restore the working page and the loaded tasks group") + } + if strings.Contains(systemOf(agent), "This model has no tools in codeaf") { + t.Fatal("the chat page stayed after the model changed back") + } +} + +// A PINNED PROFILE GIVES WAY TO A MODEL WITH NO TOOLS and comes back after it. +func TestAPinnedProfileGivesWayToAModelWithNoTools(t *testing.T) { + t.Setenv(promptProfileEnv, "full") + agent, _ := newTestAgent(t, &scriptedCompleter{}, func(c *Config) { + buildShippedConversation(t, c) + c.System = "" + c.Model = "tool/model" + c.SupportsParameter = toolsFor + }) + if agent.config.promptProfile() != profileFull { + t.Fatalf("pinned profile = %q", agent.config.promptProfile()) + } + if err := agent.preparePromptProfile("chat/model"); err != nil { + t.Fatal(err) + } + if !agent.config.promptProfile().chat() { + t.Fatal("a pin kept a page about tools on a model with none") + } + if err := agent.preparePromptProfile("tool/model"); err != nil { + t.Fatal(err) + } + if agent.config.promptProfile() != profileFull { + t.Fatalf("after leaving the model with no tools: %q, want the pinned full", agent.config.promptProfile()) + } +} + +// A MODEL THE CATALOG DOES NOT KNOW KEEPS THE WORKING PAGE. +func TestAnUnknownModelKeepsTheWorkingPage(t *testing.T) { + t.Setenv(promptProfileEnv, "") + agent, _ := newTestAgent(t, &scriptedCompleter{steps: []step{func(context.Context, []ai.Message) (*ai.Response, error) { + return textResponse("ok"), nil + }}}, func(c *Config) { + buildShippedConversation(t, c) + c.System = "" + c.Model = "unknown/model" + c.SupportsParameter = func(string, string) (bool, bool) { return false, false } + }) + if agent.config.promptProfile().chat() || len(agent.beltDefinitions()) == 0 { + t.Fatal("an undescribed model lost its tools") + } +} + +// systemOf is the system message the next request would carry. +func systemOf(agent *Agent) string { + agent.mu.Lock() + defer agent.mu.Unlock() + return messageContentText(agent.messages[0]) +} diff --git a/internal/session/clientdoor.go b/internal/session/clientdoor.go index 9f107c7de2..bcea4bc9eb 100644 --- a/internal/session/clientdoor.go +++ b/internal/session/clientdoor.go @@ -327,9 +327,12 @@ func (a *Agent) hasClient() bool { } a.mu.Lock() defer a.mu.Unlock() - return a.client != nil + return a.hasClientLocked() } +// hasClientLocked is [Agent.hasClient] for a caller already holding a.mu. +func (a *Agent) hasClientLocked() bool { return a.client != nil } + // modelRoutingCompleter is the only completer view of a live Agent that may // leave this file. A caller may pin any model onto it; the wrapper reads that // choice and resolves the model's account before forwarding the request. diff --git a/internal/session/compact_feedback_contract_test.go b/internal/session/compact_feedback_contract_test.go new file mode 100644 index 0000000000..ad8033c81e --- /dev/null +++ b/internal/session/compact_feedback_contract_test.go @@ -0,0 +1,73 @@ +package session + +import ( + "context" + "errors" + "fmt" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestFoldReportsWhyItsSummaryWasSkipped(t *testing.T) { + for _, tc := range []struct { + name, want string + answer func(int, []ai.Message) (*ai.Response, error) + }{ + {"error", "provider unavailable", func(int, []ai.Message) (*ai.Response, error) { return nil, errors.New("provider unavailable") }}, + {"empty", "empty or unreadable", func(int, []ai.Message) (*ai.Response, error) { return textResponse(""), nil }}, + } { + t.Run(tc.name, func(t *testing.T) { + model := &summarizer{answer: tc.answer} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + for i := 0; i < 10; i++ { + agent.mu.Lock() + agent.messages = append(agent.messages, textMessage("user", fmt.Sprintf("question %d %s", i, strings.Repeat("pasted log ", 1400))), textMessage("assistant", strings.Repeat("long answer ", 400))) + agent.mu.Unlock() + } + hub := newEventHub() + changed, err := agent.compactWithPolicy(context.Background(), hub, agent.requestedCompactPolicy()) + if err != nil || !changed { + t.Fatalf("compact = %v, %v", changed, err) + } + hint := lastCompacted(t, hub).Hint + if !strings.Contains(hint, "folded") || !strings.Contains(hint, "summary skipped: ") || !strings.Contains(hint, tc.want) { + t.Fatalf("partial pass hid the failed summary: %q", hint) + } + }) + } +} + +func TestCompactionHintOmitsEqualRoundedSizes(t *testing.T) { + hint := compactionHint(compactionPass{folded: 1}, 7_850, 7_800) + if strings.Contains(hint, "→") { + t.Fatalf("the rounded size did not change: %q", hint) + } +} + +func TestCompactDoorReportsSkippedSummaryAfterFolding(t *testing.T) { + for _, tc := range []struct { + name, want string + answer func(int, []ai.Message) (*ai.Response, error) + }{ + {"error", "provider unavailable", func(int, []ai.Message) (*ai.Response, error) { return nil, errors.New("provider unavailable") }}, + {"declined", "declined to write a summary", func(int, []ai.Message) (*ai.Response, error) { + return textResponse("I'm sorry, but I cannot help with that request."), nil + }}, + } { + t.Run(tc.name, func(t *testing.T) { + model := &summarizer{answer: tc.answer} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + for i := 0; i < 10; i++ { + agent.mu.Lock() + agent.messages = append(agent.messages, textMessage("user", fmt.Sprintf("question %d %s", i, strings.Repeat("pasted log ", 1400))), textMessage("assistant", strings.Repeat("long answer ", 400))) + agent.mu.Unlock() + } + why, ok := SummarySkippedWhy(agent.Compact(context.Background())) + if !ok || !strings.Contains(why, tc.want) { + t.Fatalf("/compact hid the skipped summary: why=%q ok=%v", why, ok) + } + }) + } +} diff --git a/internal/session/compact_history_contract_test.go b/internal/session/compact_history_contract_test.go new file mode 100644 index 0000000000..69f5f12e10 --- /dev/null +++ b/internal/session/compact_history_contract_test.go @@ -0,0 +1,65 @@ +package session + +import ( + "context" + "path/filepath" + "strings" + "sync/atomic" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +// The screen joins scrollback above the live transcript's floor. A turn that +// finished during the summary must remain on that joined page after each pass. +func TestMidSummaryExchangeAppearsOnceInJoinedHistory(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var summaries atomic.Int32 + model := &summarizer{answer: func(_ int, messages []ai.Message) (*ai.Response, error) { + if strings.HasPrefix(messageText(messages[0]), "You are compacting") { + if summaries.Add(1) == 1 { + close(started) + <-release + } + return textResponse(summaryWords), nil + } + return textResponse("MIDFLIGHT ANSWER"), nil + }} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 65_536 + c.SessionFile = filepath.Join(t.TempDir(), "session.jsonl") + }) + personHeavy(agent, 12, 20_000) + done := make(chan error, 1) + go func() { done <- agent.Compact(context.Background()) }() + <-started + drainTurn(t, agent, "MIDFLIGHT QUESTION") + close(release) + if err := <-done; err != nil { + t.Fatal(err) + } + checkJoinedExchange(t, agent) + personHeavy(agent, 12, 20_000) + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + checkJoinedExchange(t, agent) +} + +func checkJoinedExchange(t *testing.T, agent *Agent) { + t.Helper() + history, transcript := agent.EarlierHistory(), agent.Transcript() + joined := append(append([]DisplayEntry(nil), history.Entries...), transcript[min(history.Floor, len(transcript)):]...) + for _, word := range []string{"MIDFLIGHT QUESTION", "MIDFLIGHT ANSWER"} { + count := 0 + for _, entry := range joined { + if strings.Contains(entry.Text, word) { + count++ + } + } + if count != 1 { + t.Fatalf("%q appears %d times in joined history (floor=%d, earlier=%d, live=%d)", word, count, history.Floor, len(history.Entries), len(transcript)) + } + } +} diff --git a/internal/session/compact_meter_contract_test.go b/internal/session/compact_meter_contract_test.go new file mode 100644 index 0000000000..62c6b26bd3 --- /dev/null +++ b/internal/session/compact_meter_contract_test.go @@ -0,0 +1,52 @@ +package session + +import ( + "context" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestContextMeterAfterCompactIncludesSentToolDefinitions(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + personHeavy(agent, 8, 4_000) + belt := agent.beltTokens() + if belt == 0 { + t.Fatal("fixture has no tool definitions") + } + hub := newEventHub() + if changed, err := agent.compactWithPolicy(context.Background(), hub, agent.requestedCompactPolicy()); err != nil || !changed { + t.Fatalf("compact = %v, %v", changed, err) + } + agent.mu.Lock() + transcript := agent.transcriptTokensLocked() + agent.mu.Unlock() + if got := agent.ContextTokens(); got < belt+transcript { + t.Fatalf("post-compact meter %d omits belt %d from transcript %d", got, belt, transcript) + } + shown := agent.ContextTokens() + if hint := lastCompacted(t, hub).Hint; !strings.Contains(hint, approxTokens(shown)) { + t.Fatalf("compaction line %q does not use the post-pass meter %d", hint, shown) + } + reported := 0 + model.answer = func(_ int, messages []ai.Message) (*ai.Response, error) { + bytes := 0 + for _, message := range messages { + bytes += messageBytes(message) + } + reported = EstimateTokens(bytes) + belt + response := textResponse("yes") + response.Usage.PromptTokens = reported + return response, nil + } + for _, event := range collect(t, mustSubmit(t, agent, "one more question")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if reported == 0 || abs(reported-shown) > max(100, reported/20) { + t.Fatalf("meter after compact %d differs from next reported prompt %d", shown, reported) + } +} diff --git a/internal/session/compact_policy.go b/internal/session/compact_policy.go new file mode 100644 index 0000000000..7d15db42aa --- /dev/null +++ b/internal/session/compact_policy.go @@ -0,0 +1,178 @@ +package session + +import ( + "context" + + "github.com/Agent-Field/codeaf/internal/provider" +) + +// A requested or necessary reduction has different priorities from background +// cache maintenance. It may archive completed work in the current turn, but +// never user instructions, the system prompt, or the newest tool batch. +type compactPolicy struct { + target int + keep int + active bool + // manual is a person's /compact. It may write a summary even when the + // conversation is under every line, once the free rungs found nothing. + manual bool + // summarize lets the pass end with a summary (compact_summary.go) when the + // free rungs leave the transcript above summarizeAbove; the summary then + // aims for summarizeTo. Both are TRANSCRIPT tokens — the messages alone, + // with the belt's definitions already taken off the window's lines + // ([Agent.beltTokens]) — because the transcript is the only part a summary + // can shrink. + summarize bool + summarizeAbove int + summarizeTo int + // recovering is the pass a refused request runs: it is asked for a + // specific reclaim, so a smaller region is worth summarizing + // ([summaryWorthIt]). + recovering bool +} + +// automaticCompactPolicy is the pass the loop runs on its own: the threshold +// fires it, the fold aims below it, and a summary is written only when the free +// rungs leave the conversation above the threshold. +func (a *Agent) automaticCompactPolicy() compactPolicy { + belt := a.beltTokens() + // THE SYSTEM PAGE IS THE FIRST MESSAGE WHEN THERE IS ONE. An agent that + // has not been given it yet has nothing fixed to measure, and indexing an + // empty transcript would panic the pass that was only asking. + system := 0 + a.mu.Lock() + if len(a.messages) > 0 { + system = EstimateTokens(a.transcriptMessageBytesLocked(0)) + } + a.mu.Unlock() + line := a.compactTargetTokens() - belt + return compactPolicy{ + target: a.compactTargetTokens(), + keep: a.keepRecentTokens(), + // A summary cannot make the fixed system page or the note smaller. + summarize: line > system+summaryNoteTokens, + summarizeAbove: a.compactThreshold() - belt, + summarizeTo: line, + } +} + +// compactRecentTokens keeps a useful working tail without protecting an entire +// small-window conversation. This is a token allowance, not a summary length. +const compactRecentTokens = 4096 +const contextRecoveryAttempts = 2 + +// recoveryRoomDivisor sizes the room a refusal's recovery frees beyond what +// was missing: a thirty-second of the window, 512 tokens of a 16k window and +// 4,096 of a 128k one — enough for the next message, not a quarter of the +// conversation. +const recoveryRoomDivisor = 32 + +func (a *Agent) requestedCompactPolicy() compactPolicy { + // A person who typed /compact wants the conversation as short as it may + // be made, so a line the belt has already crossed becomes the smallest + // one there is rather than a reason to do nothing. + line := max(1, a.compactTargetTokens()-a.beltTokens()) + return compactPolicy{ + keep: min(compactRecentTokens, a.trustedWindow()/8), + active: true, + manual: true, + summarize: true, + summarizeAbove: line, + summarizeTo: line, + } +} + +// recoverContext only retries a changed request. An endpoint's explicit window +// is evidence; the failed prompt's estimated size is not a context limit. +func (a *Agent) recoverContext(ctx context.Context, hub *eventHub, err error) bool { + // A manual pass may be buying the very room this refused turn needs. + // Wait for its signal once, then send the changed request; if it did not + // shrink the transcript, run the ordinary recovery policy below. + a.mu.Lock() + beforePass := a.transcriptTokensLocked() + done := a.compactDone + compacting := a.compacting + a.mu.Unlock() + if compacting && done != nil { + waitCtx, cancel := context.WithTimeout(ctx, CompactPatience) + select { + case <-done: + case <-waitCtx.Done(): + cancel() + return false + } + cancel() + a.mu.Lock() + roomMade := a.transcriptTokensLocked() < beforePass + a.mu.Unlock() + if roomMade { + return true + } + } + failure, _ := provider.RefusalFrom(err) + calibrated := false + if failure != nil && failure.InputTokens > 0 && !failure.Local { + a.mu.Lock() + calibrated = failure.InputTokens > a.contextTokens + if calibrated { + a.contextTokens = failure.InputTokens + } + a.mu.Unlock() + } + if failure != nil && failure.ContextLimit > 0 { + a.servedWindow.Store(int64(failure.ContextLimit)) + } + policy := a.requestedCompactPolicy() + policy.manual = false + a.mu.Lock() + before := a.transcriptTokensLocked() + a.mu.Unlock() + // WITH NO FIGURES, A QUARTER. A refusal that says neither the limit nor the + // size gives no measure of what is missing, and a generous guess costs one + // pass where a stingy one costs a second refusal. + reclaim := max(1, before/4) + if failure != nil && failure.ContextLimit > 0 && failure.InputTokens > 0 { + allowed := failure.ContextLimit - failure.OutputTokens - provider.ContextSafetyTokens(failure.ContextLimit) + // WITH FIGURES, WHAT IS MISSING AND A LITTLE ROOM. The quarter used to + // be a floor here too, and it made a request 78 tokens over reclaim 1,800 + // — a summary of the very answer the person was asking about, when + // what was missing was a sentence. The room ([recoveryRoomDivisor]) + // is what keeps the next turn's message from being refused at once. + // A refusal whose figures say it already fits leaves the quarter + // standing, because then the estimate is what is wrong. + if need := failure.InputTokens - allowed; need > 0 { + reclaim = min(before, need+failure.ContextLimit/recoveryRoomDivisor) + } + } + policy.target = max(1, before-reclaim) + // A refused request needs this line reached, by a summary if the free + // rungs cannot get there. The target is already in transcript tokens: the + // reclaim above was measured against the refused request as a whole. + policy.summarizeAbove = policy.target + policy.summarizeTo = policy.target + policy.recovering = true + changed, compactErr := a.compactWithPolicy(ctx, hub, policy) + a.mu.Lock() + smaller := a.transcriptTokensLocked() < before + a.mu.Unlock() + return (changed && compactErr == nil && smaller) || calibrated || (failure != nil && failure.BudgetChanged) +} + +// transcriptMessageBytesLocked includes the working carried beside assistant +// messages. Folding only visible prose while replaying its reasoning can leave +// the provider request large after the local meter claims it shrank. +func (a *Agent) transcriptMessageBytesLocked(index int) int { + size := messageBytes(a.messages[index]) + if index < len(a.messageReasoning) { + working := a.messageReasoning[index] + size += len(working.Text) + len(working.Details) + } + return size +} +func (a *Agent) transcriptTokensLocked() int { + total := 0 + for i := range a.messages { + total += a.transcriptMessageBytesLocked(i) + } + return EstimateTokens(total) +} diff --git a/internal/session/compact_policy_test.go b/internal/session/compact_policy_test.go new file mode 100644 index 0000000000..2e2e6a931d --- /dev/null +++ b/internal/session/compact_policy_test.go @@ -0,0 +1,266 @@ +package session + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/exec/bare" + "github.com/Agent-Field/codeaf/internal/provider" +) + +// /compact shortens a conversation already under the automatic target, as far +// as it goes: the fold takes the old answers and the summary the older +// questions, and the three most recent stay word for word. +func TestManualCompactionReducesHistoryBelowAutomaticTarget(t *testing.T) { + agent, _ := newTestAgent(t, &summarizer{}, func(c *Config) { c.ContextWindow = 131072; c.SessionFile = filepath.Join(t.TempDir(), "session.jsonl") }) + grownTranscript(agent, 30, 8000) + before := estimate(agent) + if before >= agent.compactTargetTokens() { + t.Fatal("fixture crossed automatic target") + } + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if after := estimate(agent); after >= before/2 { + t.Fatalf("manual compaction barely changed %d to %d", before, after) + } + messages := liveTranscript(agent) + for i := 28; i <= 30; i++ { + if !holdsText(messages, fmt.Sprintf("question %d", i)) { + t.Fatalf("lost recent question %d", i) + } + } +} + +func TestEmergencyCompactionCanArchiveCompletedActiveBatches(t *testing.T) { + agent, _ := newTestAgent(t, &refusingCompleter{t: t}, func(c *Config) { c.ContextWindow = 40960; c.SessionFile = filepath.Join(t.TempDir(), "session.jsonl") }) + agent.mu.Lock() + agent.running = true + agent.turnFloor = 1 + agent.messages = append(agent.messages, textMessage("user", "keep my instructions")) + for i := 0; i < 16; i++ { + id := fmt.Sprintf("call-%d", i) + agent.messages = append(agent.messages, ai.Message{Role: "assistant", ToolCalls: []ai.ToolCall{{ID: id, Function: ai.ToolCallFunction{Name: "edit", Arguments: `{"path":"main.go"}`}}}}, ai.Message{Role: "tool", ToolCallID: id, Content: []ai.ContentPart{{Type: "text", Text: strings.Repeat("edited content ", 600)}}}) + } + agent.mu.Unlock() + defer func() { agent.mu.Lock(); agent.running = false; agent.mu.Unlock() }() + before := estimate(agent) + if !agent.recoverContext(context.Background(), nil, &provider.APIError{Overflow: true, ContextLimit: 40960, InputTokens: before, OutputTokens: 14746}) { + t.Fatal("no recovery") + } + after := estimate(agent) + if after >= before { + t.Fatalf("did not shrink: %d -> %d", before, after) + } + messages := liveTranscript(agent) + assertPaired(t, messages) + if !holdsText(messages, "keep my instructions") { + t.Fatal("lost user message") + } + if messages[len(messages)-1].ToolCallID != "call-15" { + t.Fatal("newest result folded") + } +} + +func TestContextRecoveryResetsAfterSuccessfulToolSteps(t *testing.T) { + refused := func(context.Context, []ai.Message) (*ai.Response, error) { + return nil, &provider.APIError{Status: 400, Overflow: true, ContextLimit: 40960, BudgetChanged: true, Message: "context length"} + } + completer := &scriptedCompleter{steps: []step{ + refused, + func(context.Context, []ai.Message) (*ai.Response, error) { + return toolResponse("one", "change", `{}`), nil + }, + refused, + func(context.Context, []ai.Message) (*ai.Response, error) { + return toolResponse("two", "change", `{}`), nil + }, + func(context.Context, []ai.Message) (*ai.Response, error) { return textResponse("done"), nil }, + }} + agent, _ := newTestAgent(t, completer, nil) + changes := 0 + agent.tools = append(agent.tools, bare.Tool{Name: "change", Description: "records a change", Schema: json.RawMessage(`{"type":"object"}`), Execute: func(context.Context, json.RawMessage) (string, bool, error) { changes++; return "changed", false, nil }}) + for _, event := range collect(t, mustSubmit(t, agent, "make both changes")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if completer.requests() != 5 || changes != 2 { + t.Fatalf("requests=%d changes=%d; completed actions must not replay", completer.requests(), changes) + } +} + +func TestContextRecoveryDoesNotLoopWithoutProgress(t *testing.T) { + completer := &scriptedCompleter{steps: []step{func(context.Context, []ai.Message) (*ai.Response, error) { + return nil, &provider.APIError{Status: 400, Overflow: true, Message: "context length"} + }}} + agent, _ := newTestAgent(t, completer, nil) + var failure error + for _, event := range collect(t, mustSubmit(t, agent, "hello")) { + if event.Kind == EventError { + failure = event.Err + } + } + if failure == nil || completer.requests() != 1 { + t.Fatalf("error=%v requests=%d", failure, completer.requests()) + } + if err := agent.Compact(context.Background()); !errors.Is(err, ErrNothingToCompact) { + t.Fatal(err) + } +} + +// This drives the real session door through the HTTP adapter. Both rejections +// have explicit endpoint windows, and real tools run between them. +func TestContextRecoveryThroughHTTPKeepsWorkingAfterSecondOverflow(t *testing.T) { + var requests atomic.Int32 + var outputCaps [5]atomic.Int64 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body struct { + MaxTokens int `json:"max_tokens"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Error(err) + } + n := int(requests.Add(1)) + if n <= len(outputCaps) { + outputCaps[n-1].Store(int64(body.MaxTokens)) + } + w.Header().Set("Content-Type", "application/json") + switch n { + case 1, 3: + limit := 40960 + if n == 3 { + limit = 32768 + } + w.WriteHeader(400) + fmt.Fprintf(w, `{"error":{"code":"context_length_exceeded","message":"This model's maximum context length is %d tokens. However, you requested 14746 output tokens and your prompt contains at least 26215 input tokens.","metadata":{"provider_name":"SmallEndpoint"}}}`, limit) + case 2, 4: + fmt.Fprintf(w, `{"model":"budget/http","choices":[{"index":0,"finish_reason":"tool_calls","message":{"role":"assistant","tool_calls":[{"id":"change-%d","type":"function","function":{"name":"change","arguments":"{}"}}]}}],"usage":{"prompt_tokens":26215,"completion_tokens":10}}`, n) + default: + fmt.Fprint(w, `{"model":"budget/http","choices":[{"index":0,"finish_reason":"stop","message":{"role":"assistant","content":"finished both changes"}}],"usage":{"prompt_tokens":26215,"completion_tokens":10}}`) + } + })) + defer server.Close() + client, err := provider.NewClient(provider.Config{APIKey: "fixture", BaseURL: server.URL, Model: "budget/http", Direct: true}) + if err != nil { + t.Fatal(err) + } + agent, _ := newTestAgent(t, client, func(c *Config) { c.Model = "budget/http"; c.ContextWindow = 131072 }) + changes := 0 + agent.tools = append(agent.tools, bare.Tool{Name: "change", Description: "records one action", Schema: json.RawMessage(`{"type":"object"}`), Execute: func(context.Context, json.RawMessage) (string, bool, error) { changes++; return "changed", false, nil }}) + for _, event := range collect(t, mustSubmit(t, agent, "make both changes")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if requests.Load() != 5 || changes != 2 { + t.Fatalf("requests=%d actions=%d", requests.Load(), changes) + } + if outputCaps[1].Load() <= 0 || 26215+outputCaps[1].Load()+int64(provider.ContextSafetyTokens(40960)) > 40960 { + t.Fatalf("first repaired allowance=%d", outputCaps[1].Load()) + } + if outputCaps[3].Load() <= 0 || 26215+outputCaps[3].Load()+int64(provider.ContextSafetyTokens(32768)) > 32768 { + t.Fatalf("second repaired allowance=%d", outputCaps[3].Load()) + } +} + +func TestContextRecoveryBoundsRepeatedChangedLimitClaims(t *testing.T) { + refused := func(context.Context, []ai.Message) (*ai.Response, error) { + return nil, &provider.APIError{Status: 400, Overflow: true, BudgetChanged: true, Message: "context length"} + } + completer := &scriptedCompleter{steps: []step{refused, refused, refused, refused}} + agent, _ := newTestAgent(t, completer, nil) + var failure error + for _, event := range collect(t, mustSubmit(t, agent, "hello")) { + if event.Kind == EventError { + failure = event.Err + } + } + if failure == nil || completer.requests() != 1+contextRecoveryAttempts { + t.Fatalf("error=%v requests=%d", failure, completer.requests()) + } +} + +func TestLeanSmallWindowFirstRequestReachesHTTP(t *testing.T) { + t.Setenv(promptProfileEnv, "lean") + var calls atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body struct { + Messages []json.RawMessage `json:"messages"` + Tools []json.RawMessage `json:"tools"` + MaxTokens int `json:"max_tokens"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Error(err) + } + calls.Add(1) + weight := 0 + for _, message := range body.Messages { + weight += len(message) + } + for _, tool := range body.Tools { + weight += len(tool) + } + input := (weight + 3) / 4 + t.Logf("first request: input=%d output=%d tools=%d", input, body.MaxTokens, len(body.Tools)) + if input < 10000 || len(body.Tools) < 10 { + t.Error("fixture lost the shipping prompt or tool belt") + } + if body.MaxTokens < 512 || input+body.MaxTokens+provider.ContextSafetyTokens(16385) > 16385 { + t.Error("request budget does not fit") + } + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"choices":[{"message":{"role":"assistant","content":"A short answer fits."},"finish_reason":"stop"}]}`) + })) + defer server.Close() + client, err := provider.NewClient(provider.Config{APIKey: "fixture", BaseURL: server.URL, Model: "budget/lean-first", Direct: true}) + if err != nil { + t.Fatal(err) + } + agent, _ := newTestAgent(t, client, func(c *Config) { + buildShippedConversation(t, c) + c.System = "" + c.Model = "budget/lean-first" + c.ContextWindow = 16385 + // Include repository instructions; the old prefix weighing omitted them. + if err := os.WriteFile(filepath.Join(c.Workspace, "AGENTS.md"), []byte(strings.Repeat("Follow project conventions.\n", 200)), 0600); err != nil { + t.Fatal(err) + } + }) + for _, event := range collect(t, mustSubmit(t, agent, "Explain the conjecture.")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if calls.Load() != 1 { + t.Fatalf("HTTP requests = %d", calls.Load()) + } +} + +func TestAutomaticNoOpCompactionEmitsNoSeam(t *testing.T) { + agent, _ := newTestAgent(t, &refusingCompleter{t: t}, nil) + hub := newEventHub() + events := hub.subscribe() + if _, err := agent.compact(context.Background(), hub); !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("compact error=%v", err) + } + hub.close() + for _, event := range collect(t, events) { + if event.Kind == EventCompacting || event.Kind == EventCompacted { + t.Fatal("automatic no-op emitted a compaction seam") + } + } + if err := agent.Compact(context.Background()); !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("manual no-op lost its feedback: %v", err) + } +} diff --git a/internal/session/compact_r2_contract_test.go b/internal/session/compact_r2_contract_test.go new file mode 100644 index 0000000000..9a91ebe829 --- /dev/null +++ b/internal/session/compact_r2_contract_test.go @@ -0,0 +1,79 @@ +package session + +import ( + "context" + "fmt" + "strings" + "testing" + + "github.com/Agent-Field/codeaf/internal/provider" +) + +func TestManualSummaryKeepsThreeWhenNoCutCanReachTarget(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 32_768 + c.System = strings.Repeat("fixed system instruction ", 3500) + }) + personHeavy(agent, 8, 4_000) + if fixed := EstimateTokens(agent.transcriptMessageBytesLocked(0)) + agent.beltTokens(); fixed <= agent.compactTargetTokens() { + t.Fatalf("fixed prefix %d does not exceed target %d", fixed, agent.compactTargetTokens()) + } + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if model.calls() != 1 { + t.Fatalf("summary calls = %d, want one", model.calls()) + } + messages := liveTranscript(agent) + for turn := 6; turn <= 8; turn++ { + if !holdsText(messages, fmt.Sprintf("question %d:", turn)) { + t.Fatalf("last three messages lost question %d", turn) + } + } + if holdsText(messages, "question 5:") { + t.Fatal("older question was not summarized") + } +} + +func TestAutomaticSummaryKeepsThreeWhenNoCutCanReachTarget(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 32_768 + c.System = strings.Repeat("fixed system instruction ", 900) + }) + personHeavy(agent, 8, 4_000) + policy := agent.automaticCompactPolicy() + if !policy.summarize { + t.Fatal("fixture has no reachable summary call") + } + if changed, err := agent.compactWithPolicy(context.Background(), nil, policy); !changed || err != nil { + t.Fatalf("automatic compact = %v, %v", changed, err) + } + if model.calls() != 1 { + t.Fatalf("automatic summary calls = %d, want one", model.calls()) + } + messages := liveTranscript(agent) + for turn := 6; turn <= 8; turn++ { + if !holdsText(messages, fmt.Sprintf("question %d:", turn)) { + t.Fatalf("automatic pass lost recent question %d", turn) + } + } + if holdsText(messages, "question 5:") { + t.Fatal("automatic pass did not summarize the older question") + } +} + +func TestRefusedRequestCanKeepFewerThanThreeWhenRoomIsMissing(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 32_768 }) + personHeavy(agent, 8, 4_000) + failure := &provider.APIError{Status: 400, Overflow: true, ContextLimit: 32_768, InputTokens: 40_000, OutputTokens: 512} + if !agent.recoverContext(context.Background(), nil, failure) { + t.Fatal("refused request was not shortened") + } + messages := liveTranscript(agent) + if !holdsText(messages, "question 8:") || holdsText(messages, "question 6:") { + t.Fatal("recovery did not keep the latest message and reclaim older ones") + } +} diff --git a/internal/session/compact_r4_contract_test.go b/internal/session/compact_r4_contract_test.go new file mode 100644 index 0000000000..6d93026101 --- /dev/null +++ b/internal/session/compact_r4_contract_test.go @@ -0,0 +1,173 @@ +package session + +import ( + "context" + "errors" + "fmt" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/provider" +) + +type ceilingSummarizer struct { + ceilings []int + answerAt int + alwaysEmpty bool +} + +func (s *ceilingSummarizer) CompleteWithMessages(_ context.Context, _ []ai.Message, options ...ai.Option) (*ai.Response, error) { + request := ai.Request{} + for _, option := range options { + if err := option(&request); err != nil { + return nil, err + } + } + ceiling := 0 + if request.MaxTokens != nil { + ceiling = *request.MaxTokens + } + s.ceilings = append(s.ceilings, ceiling) + if s.alwaysEmpty || ceiling < s.answerAt { + response := textResponse("") + response.Choices[0].FinishReason = "length" + response.Usage.CompletionTokens = ceiling + return response, nil + } + return textResponse(summaryWords), nil +} + +func TestSummaryRetriesOnceWhenThinkingConsumesTheFirstCeiling(t *testing.T) { + model := &ceilingSummarizer{answerAt: 2500} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 32_768 }) + personHeavy(agent, 8, 4_000) + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if len(model.ceilings) != 2 || model.ceilings[0] >= model.answerAt || model.ceilings[1] < model.answerAt || model.ceilings[1] >= agent.trustedWindow() { + t.Fatalf("summary ceilings = %v, want one bounded retry with thinking room", model.ceilings) + } + if summaryNotes(liveTranscript(agent)) != 1 { + t.Fatal("the retried answer was not used") + } + if used := agent.Usage(); used.Calls != 2 { + t.Fatalf("summary spend has %d calls, want both attempts", used.Calls) + } +} + +func TestSummaryNormalAnswerKeepsOneCallAndItsCeiling(t *testing.T) { + model := &ceilingSummarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 32_768 }) + personHeavy(agent, 8, 4_000) + agent.mu.Lock() + plan, why, ok := agent.planSummaryLocked(agent.requestedCompactPolicy()) + agent.mu.Unlock() + if !ok { + t.Fatalf("fixture has no summary: %s", why) + } + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if len(model.ceilings) != 1 || model.ceilings[0] != plan.ceiling { + t.Fatalf("normal summary ceilings = %v, want only %d", model.ceilings, plan.ceiling) + } +} + +func TestSummaryChunksLeaveTheListedThinkingPassItsWindowRoom(t *testing.T) { + profile := provider.ReasoningProfile{Mandatory: true, Efforts: []provider.Effort{provider.EffortHigh, "xhigh"}, Default: "xhigh"} + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 16_384 + c.ReasoningProfile = func(string) (provider.ReasoningProfile, bool) { return profile, true } + }) + personHeavy(agent, 10, 20_000) + agent.mu.Lock() + plan, why, ok := agent.planSummaryLocked(agent.requestedCompactPolicy()) + agent.mu.Unlock() + if !ok { + t.Fatalf("fixture has no summary: %s", why) + } + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if model.calls() == 0 { + t.Fatal("the summary was not requested") + } + reserve := provider.SummaryOutputReserve(profile, plan.ceiling) + for index, ask := range model.asks { + prompt := EstimateTokens(len(messageText(ask[0])) + len(messageText(ask[1]))) + if prompt+reserve+provider.ContextSafetyTokens(plan.window) > plan.window { + t.Fatalf("summary chunk %d uses %d input tokens and leaves less than %d output tokens in the %d-token window", index, prompt, reserve, plan.window) + } + } +} + +func TestSummaryStillEmptyAfterRetryKeepsTheExistingReason(t *testing.T) { + model := &ceilingSummarizer{alwaysEmpty: true} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 32_768 }) + personHeavy(agent, 8, 4_000) + err := agent.Compact(context.Background()) + why, ok := NothingToCompactWhy(err) + if !ok || !strings.Contains(why, "the model's summary came back empty or unreadable") { + t.Fatalf("Compact = %v, want the existing empty-summary reason", err) + } + if len(model.ceilings) != 2 { + t.Fatalf("summary calls = %d, want exactly two", len(model.ceilings)) + } + if summaryNotes(liveTranscript(agent)) != 0 { + t.Fatal("an empty answer replaced the conversation") + } +} + +func TestUnreachableLineKeepsBothRemainingPersonMessages(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 32_768 + c.System = strings.Repeat("fixed system instruction ", 3500) + }) + agent.mu.Lock() + agent.messages = append(agent.messages, + textMessage("user", summaryNote("The person established the earlier plan.", "grep or read x")), + textMessage("user", "question one: "+strings.Repeat("pasted log line ", 800)), + textMessage("assistant", "answer one"), + textMessage("user", "question two")) + agent.mu.Unlock() + if fixed := EstimateTokens(agent.transcriptMessageBytesLocked(0)) + agent.beltTokens(); fixed <= agent.compactTargetTokens() { + t.Fatalf("fixed prefix %d does not exceed target %d", fixed, agent.compactTargetTokens()) + } + err := agent.Compact(context.Background()) + if !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("Compact = %v, want too little older material", err) + } + if model.calls() != 0 { + t.Fatalf("summary calls = %d, want none", model.calls()) + } + for _, want := range []string{"question one:", "question two"} { + if !holdsText(liveTranscript(agent), want) { + t.Fatal(fmt.Sprintf("protected message %q was lost", want)) + } + } +} + +func TestRollingSummaryRequestCarriesPreviousFactsVerbatim(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + agent.mu.Lock() + agent.messages = append(agent.messages, textMessage("user", summaryNote("The vault code is PLUM-7731 and the file is /tmp/vault.txt.", "grep or read x"))) + agent.mu.Unlock() + personHeavy(agent, 8, 20_000) + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if model.calls() == 0 { + t.Fatal("the rolling summary was not requested") + } + first := model.asks[0] + if !strings.Contains(messageText(first[1]), "PLUM-7731") || !strings.Contains(messageText(first[1]), "/tmp/vault.txt") { + t.Fatal("the previous summary's facts were not sent") + } + if instruction := messageText(first[0]); !strings.Contains(instruction, "previous summary") || !strings.Contains(instruction, "word for word") || !strings.Contains(instruction, "supersedes") { + t.Fatalf("rolling summary instruction does not protect prior facts: %q", instruction) + } +} diff --git a/internal/session/compact_refusal_contract_test.go b/internal/session/compact_refusal_contract_test.go new file mode 100644 index 0000000000..f245af3a85 --- /dev/null +++ b/internal/session/compact_refusal_contract_test.go @@ -0,0 +1,55 @@ +package session + +import ( + "context" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +func TestRefusedSummaryDoesNotReplaceConversation(t *testing.T) { + for _, tc := range []struct { + name string + answer func(int, []ai.Message) (*ai.Response, error) + }{ + {"refusal prose", func(int, []ai.Message) (*ai.Response, error) { + return textResponse("I'm sorry, but I can't help with that request."), nil + }}, + {"content filter", func(int, []ai.Message) (*ai.Response, error) { + response := textResponse(summaryWords) + response.Choices[0].FinishReason = "content_filter" + return response, nil + }}, + } { + t.Run(tc.name, func(t *testing.T) { + model := &summarizer{answer: tc.answer} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + personHeavy(agent, 8, 20_000) + hub := newEventHub() + _, _ = agent.compactWithPolicy(context.Background(), hub, agent.requestedCompactPolicy()) + if summaryNotes(liveTranscript(agent)) != 0 || !holdsText(liveTranscript(agent), "question 1:") { + t.Fatal("the refusal replaced the older conversation") + } + if hint := lastCompacted(t, hub).Hint; !strings.Contains(hint, "summary skipped: the model declined to write a summary") { + t.Fatalf("the pass did not explain the refusal: %q", hint) + } + }) + } +} + +func TestRoleWordsDoNotMakeSummaryRefusalLookSubstantive(t *testing.T) { + for _, answer := range []string{ + "I'm sorry, but I cannot help this person with that request.", + "I'm sorry, but as an AI assistant I can't help with that.", + } { + model := &summarizer{answer: func(int, []ai.Message) (*ai.Response, error) { return textResponse(answer), nil }} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 65_536 }) + personHeavy(agent, 8, 20_000) + hub := newEventHub() + _, _ = agent.compactWithPolicy(context.Background(), hub, agent.requestedCompactPolicy()) + if summaryNotes(liveTranscript(agent)) != 0 || !holdsText(liveTranscript(agent), "question 1:") { + t.Fatalf("role word in refusal replaced conversation: %q", answer) + } + } +} diff --git a/internal/session/compact_spend_contract_test.go b/internal/session/compact_spend_contract_test.go new file mode 100644 index 0000000000..128fb35f9d --- /dev/null +++ b/internal/session/compact_spend_contract_test.go @@ -0,0 +1,64 @@ +package session + +import ( + "context" + "path/filepath" + "testing" + "time" + + "github.com/Agent-Field/codeaf/internal/catalog" + "github.com/Agent-Field/codeaf/internal/config" + "github.com/Agent-Field/codeaf/internal/crewroute" + "github.com/Agent-Field/codeaf/internal/router" +) + +func TestReroutedSummarySpendNamesAnsweringModel(t *testing.T) { + t.Setenv(config.APIKeyEnv, "") + profile := t.TempDir() + if err := config.WriteAPIKey(profile, "sk-or-v1-test-0123456789"); err != nil { + t.Fatal(err) + } + const chat = "z-ai/glm-5.3-flash" + const worker = "moonshotai/kimi-k3" + oldCatalog, oldHistory, oldGood := config.CrewCatalog, config.CrewRouteHistory, config.CrewLastGood + t.Cleanup(func() { + config.CrewCatalog, config.CrewRouteHistory, config.CrewLastGood = oldCatalog, oldHistory, oldGood + }) + config.CrewCatalog = func() []catalog.Model { + return []catalog.Model{ + {ID: chat, OpenWeights: true, PromptPrice: 1.5e-7, CompletionPrice: 5e-7, IntelligenceIndex: 41.8, CodingIndex: 71.5, AgenticIndex: 50.9, ArenaElo: 1348, ContextLength: 1310720, Parameters: []string{"tools"}}, + {ID: worker, OpenWeights: true, PromptPrice: 3e-6, CompletionPrice: 1.5e-5, IntelligenceIndex: 43.6, CodingIndex: 76.2, AgenticIndex: 50, ArenaElo: 1421, ContextLength: 1048576, Parameters: []string{"tools"}}, + } + } + config.CrewRouteHistory = func(string) []router.CrewRouteOutcome { + return []router.CrewRouteOutcome{{At: time.Now(), Send: chat, Provider: "openrouter", Kind: "forbidden"}} + } + config.CrewLastGood = func(string) *router.CrewRecord { return nil } + if err := config.SetCrewPin(profile, crewroute.Worker, worker); err != nil { + t.Fatal(err) + } + path := filepath.Join(t.TempDir(), "session.jsonl") + agent, _ := newTestAgent(t, &summarizer{}, func(c *Config) { + c.Model = chat + c.ProfileDir = profile + c.RouteCrew = func(config.CrewAsk) (crewroute.Decision, error) { return crewroute.Decision{}, nil } + c.ContextWindow = 65_536 + c.SessionFile = path + }) + if got := agent.healthyModel(context.Background(), purposeSummary, chat); got != worker { + t.Fatalf("fixture did not reroute the summary: %q", got) + } + personHeavy(agent, 8, 20_000) + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + for _, line := range journalUsageLines(t, path) { + if line.Role == auxRoleSummary { + if line.Model != worker { + t.Fatalf("summary spend names %q, want answering model %q", line.Model, worker) + } + return + } + } + t.Fatal("the summary wrote no spend row") +} diff --git a/internal/session/compact_summary.go b/internal/session/compact_summary.go new file mode 100644 index 0000000000..299ee89b99 --- /dev/null +++ b/internal/session/compact_summary.go @@ -0,0 +1,774 @@ +package session + +// THE SUMMARY IS THE LAST RUNG OF A COMPACTION PASS, and the only one that +// costs a model call. +// +// The two mechanical rungs (stub.go, the fold in loop.go) never touch a +// person's words and never touch the turn in hand, which is what makes them +// free and lossless. It is also what gives them a floor: a conversation whose +// weight is mostly what the person said — long questions, pasted logs, a story +// written together — or mostly the newest exchange, has nothing left that +// either rung may take, and every pass after that says "nothing to compact" +// while the window fills. Measured on 2026-09-28 against a 16k window: one +// question, one long answer and the next question were refused at 15,383 input +// tokens, and /compact found nothing eligible. +// +// So when the mechanical rungs cannot bring the conversation under the line +// the pass was asked to reach, the OLDEST part of the conversation — person and +// assistant alike — is replaced by one note the conversation's own model wrote +// about it. What never goes into a summary: +// +// - the system prompt (message 0), which is rebuilt per turn anyway; +// - the most recent person messages and everything after them +// ([summaryKeepPersonMessages]; fewer only when keeping them cannot reach +// the line), so the question being answered is always there word for word; +// - the running turn, because the region ends before the person message that +// opened it. +// +// The original lines stay above the compaction marker in the session journal, +// and the note names that file, so nothing is lost; the summary is what the +// MODEL reads from then on. A later summary folds the earlier note in rather +// than stacking a second one on it ([summaryPlan.previous]). +// +// THE CALL IS MADE WITH THE SESSION LOCK RELEASED. A summary takes seconds, +// and [Agent.Interrupt] wants the same lock; the old summarizer held it and +// could not be stopped. The region is copied before the call and compared +// after it, and a transcript that moved underneath — a rewind, anything that +// rewrote the prefix — keeps its shape and the summary is thrown away +// ([Agent.spliceSummaryLocked]). [Agent.compacting] stays set across the call, +// so no second pass can start in the gap. + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "reflect" + "strings" + "time" + "unicode" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/lane" + "github.com/Agent-Field/codeaf/internal/provider" +) + +const ( + // summaryKeepPersonMessages is how many of the person's most recent + // messages a summary prefers to leave word for word, with everything after + // the oldest of them. A region that cannot reach the pass's line while + // keeping this many keeps fewer, down to one: the message the conversation + // is currently answering is never summarized. + summaryKeepPersonMessages = 3 + + // summaryMinRegionTokens is the smallest region worth a model call. Below + // it the note, its framing and the call's cost outweigh what it could free. + summaryMinRegionTokens = 1024 + + // summaryMinRecoveryTokens is the smallest region worth a call when a + // refused request is being recovered ([summaryWorthIt]). + summaryMinRecoveryTokens = 128 + + // summaryMinAnswerTokens is the shortest summary asked for. + summaryMinAnswerTokens = 128 + + // summaryNoteTokens is the note's own framing around a summary: the opening + // sentence and the journal pointer. + summaryNoteTokens = 96 + + // summaryPromptTokens is what one summary request carries besides the + // conversation it is summarizing: the instruction, the framing of the + // user message, and the provider's message overhead. + summaryPromptTokens = 512 + + // summaryToolArgBytes and summaryToolResultBytes bound what one tool call's + // arguments and one result contribute to the text the summarizer reads. + // The summary needs to know what was done and what came back, not the + // bytes: those stay in the journal the note points to. + summaryToolArgBytes = 400 + summaryToolResultBytes = 2000 + + // summaryCallWindow bounds one summary request. The loop waits on it, so + // it is a person's wait, and it is a bound rather than a budget. + summaryCallWindow = 2 * time.Minute + + // auxRoleSummary names a summary's ledger line, so a person reading the + // session's spending can see what the compaction call cost. + auxRoleSummary = "summary" + + // purposeSummary is what the model-call log files a summary request under. + purposeSummary callPurpose = "summary" +) + +// CompactPatience is how long a surface on the far side of a connection waits +// for a /compact it asked for. A pass may now ask the model for a summary, and +// the ten seconds every other call gets is shorter than one summary request on +// a slow model (twenty seconds each on deepseek-v3.2, 2026-09-28), so the +// surface said "did not answer in time" about a pass that then landed. Two +// summary requests is the most a manual pass over one window's worth of +// conversation makes, and the minute is for the fold and the journal around +// them. +const CompactPatience = 2*summaryCallWindow + time.Minute + +// summaryNotePrefix opens every summary note. It is the same opening an older +// codeaf's summarizer wrote ([legacyCompactionNote]), so [isCompactionNote] +// already treats both as the session's own words rather than the person's. +const summaryNotePrefix = "[context compacted]" + +// summaryAnswerTokens is the largest summary asked for: a twentieth of the +// window, between 512 and 4,096 tokens. +func summaryAnswerTokens(window int) int { + return min(4096, max(512, window/20)) +} + +// summaryPlan is one summary about to be written: the region it replaces and +// everything the call needs, taken under the lock and used outside it. +type summaryPlan struct { + // end is where the region stops: messages[1:end] are summarized. + end int + // region is a copy of messages[1:end], compared against the live + // transcript before the summary is spliced in. + region []ai.Message + // previous is the body of an earlier summary note at the head of the + // region, which the new summary extends rather than repeats. + previous string + model string + window int + // answer is how long the summary is asked to be, and ceiling the most the + // request allows it to write: room for a model that overshoots the words + // it was asked for, which [Agent.spliceSummaryLocked] still refuses if the + // note ends up no smaller than what it replaces. + answer int + ceiling int + // pointer says where the original lines can be read back. + pointer string +} + +// summaryWanted says whether a pass that has already stubbed and folded still +// needs a summary. transcript is the conversation's size after those rungs. +func summaryWanted(policy compactPolicy, pass compactionPass, transcript int) bool { + // A LINE THE TRANSCRIPT CANNOT REACH IS NOT A REASON TO SUMMARIZE. When the + // tool definitions alone are past it, no summary gets the request under + // it, and a pass that summarized anyway would be back one step later + // paying for another. The refusal recovery still summarizes: it is asked + // for a reclaim it can reach. + if !policy.summarize || policy.summarizeTo <= 0 { + return false + } + if policy.manual { + // A PERSON'S /compact GOES ALL THE WAY IN ONE PASS. It asked for the + // conversation as short as it may be made ([Agent.requestedCompactPolicy]), + // so after the free rungs it summarizes whatever older conversation is + // left, however far under the automatic lines that already is; the + // region's own minimum ([summaryWorthIt]) decides whether it is worth a + // request. It used to stop once the fold got under the automatic + // target, and a person had to type /compact a second time — with + // nothing on screen saying so — to get the summary (2026-09-28). + return true + } + return transcript > policy.summarizeAbove +} + +// beltTokens is what the tool definitions add to every request, estimated from +// their encoding. The transcript estimate does not carry them, and on a small +// window they are most of what is sent. It takes the belt's own lock, so it is +// read before the session lock is. +// +// A model the catalog says takes no tools is sent none (internal/provider's +// toolless.go), so for it the belt weighs nothing. +func (a *Agent) beltTokens() int { + if a.config.SupportsParameter != nil { + a.mu.Lock() + model := a.model + a.mu.Unlock() + if supported, known := a.config.SupportsParameter(model, "tools"); known && !supported { + return 0 + } + } + definitions := a.beltDefinitions() + if len(definitions) == 0 { + return 0 + } + encoded, err := json.Marshal(definitions) + if err != nil { + return 0 + } + return EstimateTokens(len(encoded)) +} + +// planSummaryLocked chooses the region a summary replaces, and reports false +// when there is no region worth a call — with why, in the words a person's +// /compact prints (empty where there is nothing useful to say). +// +// The cuts are tried from the most kept to the least ([Agent.summaryEndsLocked]) +// and the first one that brings the conversation under the policy's line wins; +// when none does, an ordinary pass keeps the most recent three messages. +// Recovery still takes the largest worthwhile cut because a refused request +// needs every bit of room it can reclaim. +func (a *Agent) planSummaryLocked(policy compactPolicy) (summaryPlan, string, bool) { + if !a.hasClientLocked() { + return summaryPlan{}, "", false + } + window := a.trustedWindow() + if window <= 0 { + return summaryPlan{}, "", false + } + most := summaryAnswerTokens(window) + // A window too small to hold one useful chunk and its answer cannot be + // summarized into; the provider's own guard reports that case. + if summaryChunkTokens(window, most, 0) < summaryMinRegionTokens { + return summaryPlan{}, "the model's window is too small to write a summary into", false + } + persons := a.personMessagesLocked() + if len(persons) == 0 { + return summaryPlan{}, "there is no conversation to summarize yet", false + } + total := a.transcriptTokensLocked() + type cut struct{ end, tokens, answer int } + var fallback, chosen *cut + protected := min(summaryKeepPersonMessages, len(persons)) + protectedWhy := "" + // What the most a summary could have taken came to, for the sentence a + // pass that takes nothing says about itself. + freshest, noted, closed, kept := 0, false, false, 1 + for _, end := range a.summaryEndsLocked(persons) { + if !a.regionClosedLocked(end) { + continue + } + closed = true + kept = 0 + for _, person := range persons { + if person >= end { + kept++ + } + } + bytes, fresh := 0, 0 + for index := 1; index < end; index++ { + size := a.transcriptMessageBytesLocked(index) + bytes += size + if index > 1 || !strings.HasPrefix(messageContentText(a.messages[index]), summaryNotePrefix) { + fresh += size + } + } + tokens := EstimateTokens(bytes) + freshest = EstimateTokens(fresh) + noted = end > 1 && strings.HasPrefix(messageContentText(a.messages[1]), summaryNotePrefix) + if !summaryWorthIt(policy, tokens, EstimateTokens(fresh)) { + if kept == protected { + protectedWhy = summaryTooLittle(true, noted, freshest, kept) + } + // A CUT WITH NOTHING WORTH A REQUEST IS NOT A REASON TO KEEP LESS. + // Fewer of the person's messages are kept only when keeping more + // cannot get under the line; a conversation already under it has + // nothing to gain from summarizing their third-newest message, and + // a second /compact did exactly that until 2026-09-28. + if total <= policy.summarizeTo { + break + } + continue + } + candidate := &cut{end: end, tokens: tokens, answer: summaryTargetTokens(most, tokens)} + if policy.recovering || (fallback == nil && kept == protected) { + fallback = candidate + } + if total-tokens+candidate.answer+summaryNoteTokens <= policy.summarizeTo { + chosen = candidate + break + } + } + if chosen == nil { + chosen = fallback + } + if chosen == nil { + if protectedWhy != "" && !policy.recovering { + return summaryPlan{}, protectedWhy, false + } + return summaryPlan{}, summaryTooLittle(closed, noted, freshest, kept), false + } + end := chosen.end + region := make([]ai.Message, end-1) + copy(region, a.messages[1:end]) + plan := summaryPlan{ + end: end, region: region, model: a.model, window: window, + answer: chosen.answer, ceiling: min(most, 2*chosen.answer), + } + if text := messageContentText(region[0]); region[0].Role == "user" && strings.HasPrefix(text, summaryNotePrefix) { + plan.previous = summaryNoteBody(text) + } + journal, _, _ := a.file.messageLines(region[0], region[len(region)-1]) + switch { + case journal != "": + plan.pointer = "grep or read " + journal + case a.chatlog.ref(region[0]) != "": + plan.pointer = "the full record is in the store" + default: + plan.pointer = "the full record is in the session journal" + } + return plan, "", true +} + +// summaryTooLittle is what a pass says when no region was worth a summary. +func summaryTooLittle(closed, noted bool, fresh, kept int) string { + before := "your latest message" + if kept > 1 { + before = fmt.Sprintf("your last %d messages", kept) + } + switch { + case !closed: + return "the newest work is still in progress" + case fresh == 0 && noted: + return "nothing new since the last summary" + case fresh == 0: + return "there is nothing before " + before + " to summarize" + case noted: + return "only " + approxTokens(fresh) + " tokens since the last summary — too little to summarize" + default: + return "only " + approxTokens(fresh) + " tokens before " + before + " — too little to summarize" + } +} + +// summaryFailedWhy is what a pass says when the summary it asked for did not +// come back usable. +func summaryFailedWhy(ctx context.Context, err error) string { + switch { + case ctx.Err() != nil: + return "the summary was interrupted" + case errors.Is(err, errSummaryDeclined): + return "the model declined to write a summary" + case errors.Is(err, errEmptyAnswer): + return "the model's summary came back empty or unreadable" + default: + return "the model could not write a summary: " + clip(strings.TrimSpace(err.Error()), 160) + } +} + +// summaryWorthIt says whether a region is worth a model call. +// +// A REFUSED REQUEST NEEDS WHAT IT NEEDS. Recovery is bounded by its own +// allowance and is asked for a specific reclaim, so any region the model can +// say more briefly is worth it there — an earlier note included, rewritten +// shorter — however little that frees: on 2026-09-28 a turn ended refused 78 +// tokens over the window while a 740-token region sat unsummarized under a +// 1,024-token floor. +// +// EVERY OTHER PASS WANTS NEW MATERIAL. What is worth a call there is what the +// last summary has not already read ([summaryMinRegionTokens] of it): a region +// that is mostly an earlier note would be the model rewriting its own summary +// on every step, shorter and vaguer each time. +func summaryWorthIt(policy compactPolicy, tokens, fresh int) bool { + if policy.recovering { + return tokens >= summaryMinRecoveryTokens + } + return fresh >= summaryMinRegionTokens +} + +// summaryTargetTokens is how long a summary of a region is asked to be: never +// more than half the region it replaces, never more than the window's own cap, +// and never so short that it cannot say anything. +func summaryTargetTokens(most, region int) int { + return min(most, max(summaryMinAnswerTokens, region/2)) +} + +// summaryEndsLocked lists where a summary's region may end, from the cut that +// keeps the most to the one that keeps the least: +// +// 1. the person's three most recent messages and everything after them; +// 2. their two most recent; +// 3. their most recent, AND THE REPLY THAT CAME BEFORE IT — the answer a +// person's "translate it" or "keep going" is about, which a summary cannot +// stand in for (on 2026-09-28 a model asked to translate a story that had +// been summarized away translated the summary instead); +// 4. their most recent alone, which is never summarized. +func (a *Agent) summaryEndsLocked(persons []int) []int { + var ends []int + for keep := min(summaryKeepPersonMessages, len(persons)); keep >= 2; keep-- { + ends = append(ends, persons[len(persons)-keep]) + } + last := persons[len(persons)-1] + floor := 0 + if len(persons) > 1 { + floor = persons[len(persons)-2] + } + for index := last - 1; index > floor; index-- { + message := a.messages[index] + if message.Role == "assistant" && len(message.ToolCalls) == 0 && strings.TrimSpace(messageContentText(message)) != "" { + ends = append(ends, index) + break + } + } + return append(ends, last) +} + +// personMessagesLocked lists the indices of the messages the person wrote — +// the user role, less every message this package writes there. +func (a *Agent) personMessagesLocked() []int { + var persons []int + for index := 1; index < len(a.messages); index++ { + if a.messages[index].Role != "user" { + continue + } + if isCodeafNote(messageContentText(a.messages[index])) || a.file.isNote(a.messages[index]) { + continue + } + persons = append(persons, index) + } + return persons +} + +// isCodeafNote reports whether a user-role message is one this package wrote +// into the conversation rather than something the person typed, by the +// opening each is built with. +// +// IT IS WIDER THAN [isCompactionNote] AND [isVolatileNote] because a summary +// has to know which messages are the person's to keep: on 2026-09-28 the +// truncation continuation — which has no tag and reads as plain words — was +// kept as though it were the person's latest message, and the request it +// continued was summarized away in its place. +func isCodeafNote(text string) bool { + if isCompactionNote(text) || isVolatileNote(text) { + return true + } + for _, opening := range codeafNoteOpenings { + if strings.HasPrefix(text, opening) { + return true + } + } + return false +} + +// codeafNoteOpenings are the openings of the notes a turn writes into the +// conversation as the user role (checkpoint.go, inherit.go, looped.go and the +// truncation continuation in loop.go). +var codeafNoteOpenings = []string{ + truncationContinuationNote, + "[carry on] ", + checkpointChoiceLead, + "[silent] You have made ", + "[stuck] You have sent the same ", +} + +// regionClosedLocked says every tool call made in messages[1:end] is answered +// inside it. A call summarized away with its result left behind is an orphan +// every provider refuses, on this request and every one after it. +func (a *Agent) regionClosedLocked(end int) bool { + answers := toolAnswerPositions(a.messages) + for index := 1; index < end; index++ { + for call := range a.messages[index].ToolCalls { + if at, ok := answers[&a.messages[index].ToolCalls[call]]; ok && at >= end { + return false + } + } + } + return true +} + +// writeSummary asks the conversation's model for the summary, one chunk at a +// time when the region is larger than one request can carry. It is called +// with the session lock released. +func (a *Agent) writeSummary(ctx context.Context, plan summaryPlan) (string, error) { + summary := plan.previous + lines := summaryLines(plan.region) + reserve := plan.ceiling + if a.config.ReasoningProfile != nil { + if profile, known := a.config.ReasoningProfile(plan.model); known { + reserve = max(reserve, provider.SummaryOutputReserve(profile, plan.ceiling)) + } + } + reserve = min(reserve, plan.window-provider.ContextSafetyTokens(plan.window)-summaryPromptTokens-summaryMinRegionTokens) + for len(lines) > 0 { + budget := summaryChunkTokens(plan.window, reserve, EstimateTokens(len(summary))) * bytesPerToken + if budget <= 0 { + return "", errors.New("session: the summary no longer fits the window") + } + var chunk strings.Builder + taken := 0 + for taken < len(lines) { + line := lines[taken] + if chunk.Len() > 0 && chunk.Len()+len(line) > budget { + break + } + if len(line) > budget { + line = clipMiddle(line, budget) + } + chunk.WriteString(line) + taken++ + } + lines = lines[taken:] + next, err := a.askForSummary(ctx, plan, summary, chunk.String()) + if err != nil { + return "", err + } + summary = next + } + return summary, nil +} + +// summaryChunkTokens is how much conversation one summary request may carry +// once the answer, the safety allowance, the instruction and the summary so +// far are set aside. +func summaryChunkTokens(window, answer, previous int) int { + return window - answer - provider.ContextSafetyTokens(window) - summaryPromptTokens - previous +} + +// askForSummary asks the conversation's own model with no tools or stream. An +// empty length finish gets one larger request; both calls are billed. +func (a *Agent) askForSummary(ctx context.Context, plan summaryPlan, previous, chunk string) (string, error) { + ctx = provider.WithRole(provider.WithoutStream(ctx), lane.RoleAuxiliary) + ctx = provider.WithReasoningEffort(ctx, provider.EffortLow) + ctx = provider.WithContextBudget(ctx, provider.ContextBudget{Window: plan.window, Reserve: plan.ceiling}) + var ask strings.Builder + if previous != "" { + ask.WriteString("Summary so far:\n\n") + ask.WriteString(previous) + ask.WriteString("\n\nMore of the conversation, which the updated summary must also cover:\n\n") + } else { + ask.WriteString("The conversation:\n\n") + } + ask.WriteString(chunk) + ask.WriteString("\n\nWrite the summary now.") + messages := []ai.Message{ + textMessage("system", summaryInstruction(plan.answer)), + textMessage("user", ask.String()), + } + call := func(ceiling int) (*ai.Response, error) { + callCtx, cancel := context.WithTimeout(ctx, summaryCallWindow) + response, answered, err := a.completeWithNamedModel(callCtx, purposeSummary, messages, plan.model, ai.WithMaxTokens(ceiling)) + cancel() + if err != nil { + return nil, err + } + if response == nil { + return nil, errEmptyAnswer + } + a.mu.Lock() + running := a.running + a.mu.Unlock() + if running { + a.addAuxiliaryUsageAs(response, answered, 1, auxRoleSummary) + } else { + a.addDetachedUsageAs(response, answered, 1, auxRoleSummary) + } + return response, nil + } + response, err := call(plan.ceiling) + if err != nil { + return "", err + } + if provider.EmptyAtCeiling(response, plan.ceiling) { + // TWICE THE ALLOWANCE, NOT A SHARE OF THE WINDOW. The provider already + // sizes the thinking room above the answer for the level the model runs + // at, so what came back empty is the rare pass that thought past it; a + // window-sized ceiling on a million-token model would ask endpoints for + // more completion than any of them serves and be refused instead. + retryCeiling := min(plan.window-provider.ContextSafetyTokens(plan.window)-summaryPromptTokens, 2*plan.ceiling) + if retryCeiling > plan.ceiling { + response, err = call(retryCeiling) + if err != nil { + return "", err + } + } + } + return summaryAnswer(response, chunk, plan.ceiling) +} + +func summaryAnswer(response *ai.Response, chunk string, ceiling int) (string, error) { + text := strings.TrimSpace(response.Text()) + if strings.EqualFold(strings.TrimSpace(provider.FinishReason(response)), "content_filter") || summaryRefusalWithoutSubstance(text, chunk) { + return "", errSummaryDeclined + } + // An empty answer, machine markup and a model repeating itself are the + // same failure here: nothing came back that could stand in for the region. + if !briefIsProse(text) || briefRepeats(text) { + return "", errEmptyAnswer + } + return clip(text, ceiling*bytesPerToken), nil +} + +var errSummaryDeclined = errors.New("session: the model declined to write a summary") + +// A short refusal opening is not a summary when it names nothing from the +// region. Generic refusal words are ignored so a shared "request" or "help" +// cannot make an otherwise empty refusal look like conversation substance. +func summaryRefusalWithoutSubstance(answer, region string) bool { + if len(answer) > 512 { + return false + } + lower := strings.ToLower(strings.TrimSpace(answer)) + opening := false + for _, prefix := range []string{"i'm sorry", "i’m sorry", "i am sorry", "sorry,", "i can't", "i can’t", "i cannot", "i'm unable", "i am unable"} { + if strings.HasPrefix(lower, prefix) { + opening = true + break + } + } + if !opening { + return false + } + words := func(text string) []string { + return strings.FieldsFunc(strings.ToLower(text), func(r rune) bool { return !unicode.IsLetter(r) && !unicode.IsNumber(r) && r != '_' }) + } + generic := map[string]bool{"sorry": true, "cannot": true, "could": true, "would": true, "help": true, "request": true, "content": true, "policy": true, "provide": true, "assist": true, "information": true, "about": true, "that": true, "this": true, "with": true, "your": true, "their": true, "there": true, "because": true, "unable": true, "person": true, "assistant": true, "called": true, "result": true, "tool": true, "attachment": true, "shown": true, "middle": true, "omitted": true, "context": true, "compacted": true, "folded": true} + regionWords := make(map[string]bool) + for _, word := range words(region) { + if len(word) >= 5 && !generic[word] { + regionWords[word] = true + } + } + for _, word := range words(lower) { + if len(word) >= 5 && !generic[word] && regionWords[word] { + return false + } + } + return true +} + +// summaryInstruction is the summarizer's system message. +func summaryInstruction(answer int) string { + return fmt.Sprintf(`You are compacting a conversation between a person and an AI assistant so that it fits in the assistant's context window. Your summary replaces the messages you are shown, and the assistant will continue from it as if it had read them. + +Cover, in this order of importance: +- what the person asked for, and every constraint, preference or correction they gave (quote their exact words where the wording matters); +- decisions made and the reasons for them; +- what was done: files created or changed, commands run and their outcomes, errors hit and how they were resolved; +- facts learned about the project or the world that the work depends on; +- what is unfinished or still open, and what was about to happen next. + +Carry every specific fact, name, number, code, path and decision from the previous summary forward word for word unless the newer conversation explicitly supersedes it. Keep file paths, names, identifiers, numbers and error messages exact. Drop pleasantries, false starts and anything later superseded. Copy any line that begins with "[folded" verbatim: it points at the full record. Write in the third person ("the person asked", "the assistant changed"), as plain prose and short lists, with no preamble. Use at most about %d words.`, answer*3/4) +} + +// summaryLines renders the region as the text the summarizer reads, one entry +// per message. An earlier summary note is skipped here: its body travels as +// the summary so far ([summaryPlan.previous]). +func summaryLines(region []ai.Message) []string { + names := make(map[string]string, 8) + lines := make([]string, 0, len(region)) + for index, message := range region { + text := strings.TrimSpace(messageContentText(message)) + switch message.Role { + case "user": + if index == 0 && strings.HasPrefix(text, summaryNotePrefix) { + continue + } + if isVolatileNote(text) { + continue + } + if isCompactionNote(text) { + lines = append(lines, text+"\n\n") + continue + } + lines = append(lines, "PERSON: "+text+summaryImages(message)+"\n\n") + case "assistant": + var line strings.Builder + if text != "" { + line.WriteString("ASSISTANT: " + text + "\n") + } + for _, call := range message.ToolCalls { + names[call.ID] = call.Function.Name + line.WriteString("ASSISTANT CALLED " + call.Function.Name + " " + + clipMiddle(call.Function.Arguments, summaryToolArgBytes) + "\n") + } + if line.Len() > 0 { + lines = append(lines, line.String()+"\n") + } + case "tool": + name := names[message.ToolCallID] + if name == "" { + name = "tool" + } + lines = append(lines, "RESULT OF "+name+": "+clipMiddle(text, summaryToolResultBytes)+"\n\n") + } + } + return lines +} + +// summaryImages says a message carried pictures, which the summarizer is not +// shown. +func summaryImages(message ai.Message) string { + images := 0 + for _, part := range message.Content { + if part.Type != "text" { + images++ + } + } + if images == 0 { + return "" + } + return fmt.Sprintf(" [%d attachment%s not shown]", images, plural(images)) +} + +// clipMiddle keeps the start and the end of a long text, which is where a +// command's purpose and its outcome usually are. +func clipMiddle(text string, limit int) string { + if len(text) <= limit { + return text + } + const gap = "\n… [middle omitted] …\n" + if limit <= len(gap)+2 { + return clip(text, limit) + } + head := (limit - len(gap)) / 2 + tail := limit - len(gap) - head + for head > 0 && !utf8RuneStart(text[head]) { + head-- + } + start := len(text) - tail + for start < len(text) && !utf8RuneStart(text[start]) { + start++ + } + return text[:head] + gap + text[start:] +} + +// summaryNote is the user-role message a summary stands in the conversation +// as. User role because it is context handed TO the model, and marked in plain +// words because a model that mistakes a summary for a transcript will answer +// things inside it again. +func summaryNote(summary, pointer string) string { + return summaryNotePrefix + " The earlier part of this conversation was summarized to fit the " + + "context window. This note is that summary — a record, not something either of us just said. " + + "The original messages are not lost: " + pointer + ".\n\n" + summary +} + +// summaryNoteBody is the summary a note carries, without its framing. +func summaryNoteBody(note string) string { + if _, body, found := strings.Cut(note, "\n\n"); found { + return strings.TrimSpace(body) + } + return "" +} + +// spliceSummaryLocked replaces the planned region with the summary note, if +// the region is still exactly what was summarized. It reports how many +// messages the note replaced, and zero with the reason when nothing changed. +func (a *Agent) spliceSummaryLocked(plan summaryPlan, summary string) (int, string) { + if len(a.messages) < plan.end || !reflect.DeepEqual(a.messages[1:plan.end], plan.region) { + return 0, "the conversation changed while the summary was being written" + } + note := textMessage("user", summaryNote(summary, plan.pointer)) + replaced := 0 + for index := 1; index < plan.end; index++ { + replaced += a.transcriptMessageBytesLocked(index) + } + // A summary that is not smaller than what it replaces is refused, for the + // fold's reason: a pass that makes the conversation heavier is not one. + if messageBytes(note) >= replaced { + return 0, "the summary came out no shorter than what it would replace" + } + a.alignReasoningLocked() + rebuilt := make([]ai.Message, 0, len(a.messages)-plan.end+2) + rebuiltReasoning := make([]provider.MessageReasoning, 0, cap(rebuilt)) + rebuilt = append(rebuilt, a.messages[0], note) + rebuiltReasoning = append(rebuiltReasoning, a.messageReasoning[0], provider.MessageReasoning{}) + rebuilt = append(rebuilt, a.messages[plan.end:]...) + rebuiltReasoning = append(rebuiltReasoning, a.messageReasoning[plan.end:]...) + // THE RUNNING TURN'S FLOOR MOVES WITH THE REBUILD, as it does for a fold: + // the region ends before the turn's opening message, so everything from + // the floor on shifts down by the messages the note replaced. + if a.turnFloor >= plan.end { + a.turnFloor -= plan.end - 2 + } else if a.turnFloor > 1 { + a.turnFloor = 2 + } + a.messages = rebuilt + a.messageReasoning = rebuiltReasoning + return plan.end - 1, "" +} diff --git a/internal/session/compact_summary_test.go b/internal/session/compact_summary_test.go new file mode 100644 index 0000000000..9be43a4383 --- /dev/null +++ b/internal/session/compact_summary_test.go @@ -0,0 +1,604 @@ +package session + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/provider" +) + +// summaryWords is a summary the prose check accepts. +const summaryWords = "The person asked for a long story about a watchmaker and the assistant wrote it." + +// summarizer answers every request with a summary and keeps what it was sent. +type summarizer struct { + mu sync.Mutex + asks [][]ai.Message + answer func(call int, messages []ai.Message) (*ai.Response, error) + requests int +} + +func (s *summarizer) CompleteWithMessages(_ context.Context, messages []ai.Message, _ ...ai.Option) (*ai.Response, error) { + s.mu.Lock() + s.asks = append(s.asks, append([]ai.Message(nil), messages...)) + s.requests++ + call := s.requests + answer := s.answer + s.mu.Unlock() + if answer != nil { + return answer(call, messages) + } + return textResponse(fmt.Sprintf("%s (summary %d)", summaryWords, call)), nil +} + +func (s *summarizer) calls() int { + s.mu.Lock() + defer s.mu.Unlock() + return s.requests +} + +// personHeavy appends conversation whose weight is the person's own words: +// long questions, short answers, nothing a fold or a stub may take. +func personHeavy(agent *Agent, turns, bytes int) { + agent.mu.Lock() + defer agent.mu.Unlock() + for turn := 1; turn <= turns; turn++ { + agent.messages = append(agent.messages, + textMessage("user", fmt.Sprintf("question %d: %s", turn, strings.Repeat("pasted log line ", bytes/16))), + textMessage("assistant", fmt.Sprintf("answer %d", turn))) + } +} + +func summaryNotes(messages []ai.Message) int { + notes := 0 + for _, message := range messages { + if message.Role == "user" && strings.HasPrefix(messageText(message), summaryNotePrefix) { + notes++ + } + } + return notes +} + +// THE DEAD END ENDS IN A SUMMARY. A conversation that is mostly the person's +// own words has nothing the free rungs may take; the pass used to report +// nothing to compact while the window filled. +func TestAPassTheFreeRungsCannotShrinkEndsWithASummary(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { + config.ContextWindow = 65_536 + config.SessionFile = filepath.Join(t.TempDir(), "session.jsonl") + }) + personHeavy(agent, 12, 20_000) + before := estimate(agent) + if before <= agent.compactThreshold() { + t.Fatalf("fixture %d is under the threshold %d", before, agent.compactThreshold()) + } + + hub := newEventHub() + changed, err := agent.compact(context.Background(), hub) + if err != nil || !changed { + t.Fatalf("compact = %v, %v; want a pass that summarized", changed, err) + } + if model.calls() != 1 { + t.Fatalf("summary requests = %d, want 1", model.calls()) + } + messages := liveTranscript(agent) + if summaryNotes(messages) != 1 || !strings.HasPrefix(messageText(messages[1]), summaryNotePrefix) { + t.Fatalf("the summary note is not where the region was: %q", messageText(messages[1])) + } + if !strings.Contains(messageText(messages[1]), summaryWords) { + t.Fatal("the note does not carry what the model wrote") + } + // THE MOST RECENT PERSON MESSAGES STAY WORD FOR WORD, with what follows. + for turn := 10; turn <= 12; turn++ { + if !holdsText(messages, fmt.Sprintf("question %d:", turn)) || !holdsText(messages, fmt.Sprintf("answer %d", turn)) { + t.Fatalf("recent turn %d was summarized", turn) + } + } + if holdsText(messages, "question 1:") { + t.Fatal("the oldest question survived the summary") + } + if after := estimate(agent); after >= agent.compactTargetTokens() { + t.Fatalf("estimate after the summary = %d, want under the target %d", after, agent.compactTargetTokens()) + } + // The summarizer read the region as text, with no tools to call. + ask := model.asks[0] + if len(ask) != 2 || ask[0].Role != "system" || !strings.Contains(messageText(ask[1]), "PERSON: question 1:") { + t.Fatalf("summary request = %+v", ask) + } + if strings.Contains(messageText(ask[1]), "question 10:") { + t.Fatal("a kept message was sent to be summarized") + } + // THE COUNT DEPENDS ON THE FOLD, AND THE FOLD ON THE JOURNAL'S PATH. The + // eight one-line answers fold only when the marker naming the journal is + // shorter than they are: a short temporary path (Linux's /tmp) folds them + // first and summarizes the rest, a long one (macOS's /var/folders) leaves + // them to the summary. Either way every older message is gone into one of + // the two, which is what the hint must say. + event := lastCompacted(t, hub) + folded, summarized := hintCount(event.Hint, "folded"), hintCount(event.Hint, "summarized") + // The surface decides whether the line stands by this field, never by the + // hint's words (internal/tui3's workfold.go). + if event.Summarized != summarized { + t.Fatalf("event.Summarized = %d, want the %d the hint reports", event.Summarized, summarized) + } + if summarized == 0 || folded+summarized < 18 { + t.Fatalf("hint = %q; want a summary, with fold and summary covering the 18 older messages", event.Hint) + } +} + +// A FAILED SUMMARY CHANGES NOTHING. The pass reports what the free rungs did, +// which here is nothing, and the conversation keeps every word. +func TestAFailedSummaryLeavesTheConversationAlone(t *testing.T) { + model := &summarizer{answer: func(int, []ai.Message) (*ai.Response, error) { + return nil, errors.New("provider unavailable") + }} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 65_536 }) + personHeavy(agent, 12, 20_000) + before := liveTranscript(agent) + + agent.compact(context.Background(), newEventHub()) + after := liveTranscript(agent) + if summaryNotes(after) != 0 { + t.Fatal("a failed summary left a note") + } + // The fold may still take the short answers; every question stays. + for turn := 1; turn <= 12; turn++ { + if !holdsText(after, fmt.Sprintf("question %d:", turn)) { + t.Fatalf("question %d was lost to a summary that failed (%d messages, was %d)", turn, len(after), len(before)) + } + } +} + +// AN ANSWER THAT IS NOT A SUMMARY IS REFUSED like a failed call. +func TestASummaryThatIsNotProseIsRefused(t *testing.T) { + model := &summarizer{answer: func(int, []ai.Message) (*ai.Response, error) { + return textResponse("<|tool_call|>"), nil + }} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 65_536 }) + personHeavy(agent, 12, 20_000) + agent.compact(context.Background(), newEventHub()) + messages := liveTranscript(agent) + if summaryNotes(messages) != 0 || !holdsText(messages, "question 1:") { + t.Fatal("markup was spliced in as a summary") + } +} + +// THE CALL IS MADE WITHOUT THE LOCK, so the transcript can move under it. A +// region that is no longer what was summarized keeps its shape. +func TestASummaryIsDroppedWhenTheConversationMovedUnderIt(t *testing.T) { + var agent *Agent + model := &summarizer{answer: func(int, []ai.Message) (*ai.Response, error) { + agent.mu.Lock() + agent.messages[1] = textMessage("user", "question 1: rewritten while the summary was written") + agent.mu.Unlock() + return textResponse(summaryWords), nil + }} + agent, _ = newTestAgent(t, model, func(config *Config) { config.ContextWindow = 65_536 }) + personHeavy(agent, 12, 20_000) + agent.compact(context.Background(), newEventHub()) + messages := liveTranscript(agent) + if summaryNotes(messages) != 0 || !holdsText(messages, "rewritten while the summary was written") { + t.Fatal("a summary of a region that changed was spliced over the change") + } +} + +// A SECOND SUMMARY EXTENDS THE FIRST rather than stacking a second note, and +// it is handed the first as the summary so far. +func TestASecondSummaryFoldsTheFirstIn(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 65_536 }) + personHeavy(agent, 12, 20_000) + if _, err := agent.compact(context.Background(), newEventHub()); err != nil { + t.Fatal(err) + } + personHeavy(agent, 12, 20_000) + if _, err := agent.compact(context.Background(), newEventHub()); err != nil { + t.Fatal(err) + } + // The second region may take more than one request; the first of them is + // the one that has to be handed the earlier summary. + if model.calls() < 2 { + t.Fatalf("summary requests = %d, want at least 2", model.calls()) + } + messages := liveTranscript(agent) + if summaryNotes(messages) != 1 { + t.Fatalf("summary notes = %d, want the one rolling note", summaryNotes(messages)) + } + second := messageText(model.asks[1][1]) + if !strings.Contains(second, "Summary so far:") || !strings.Contains(second, "(summary 1)") { + t.Fatalf("the second summary was not handed the first: %.300q", second) + } + if !strings.Contains(messageText(messages[1]), fmt.Sprintf("(summary %d)", model.calls())) { + t.Fatal("the note does not carry the newest summary") + } +} + +// A NOTE ALONE IS NOT WORTH SUMMARIZING AGAIN. With nothing new above the kept +// messages, the pass must not pay the model to rewrite its own summary. +func TestAnEarlierSummaryAloneIsNotSummarizedAgain(t *testing.T) { + agent, _ := newTestAgent(t, &refusingCompleter{t: t}, func(config *Config) { config.ContextWindow = 65_536 }) + agent.mu.Lock() + agent.messages = append(agent.messages, + textMessage("user", summaryNote(strings.Repeat("an earlier summary sentence. ", 400), "grep or read x"))) + agent.mu.Unlock() + personHeavy(agent, 1, 200_000) + if _, err := agent.compact(context.Background(), newEventHub()); !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("compact = %v, want ErrNothingToCompact", err) + } +} + +// A REGION LARGER THAN ONE REQUEST IS SUMMARIZED IN CHUNKS, each handed the +// summary so far, and no request is larger than the window allows. +func TestALargeRegionIsSummarizedInChunks(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 16_384 }) + personHeavy(agent, 10, 24_000) + if err := agent.Compact(context.Background()); err != nil { + t.Fatal(err) + } + if model.calls() < 2 { + t.Fatalf("summary requests = %d, want the region split", model.calls()) + } + limit := (16_384 - summaryAnswerTokens(16_384) - provider.ContextSafetyTokens(16_384)) * bytesPerToken + for index, ask := range model.asks { + size := len(messageText(ask[0])) + len(messageText(ask[1])) + if size > limit { + t.Fatalf("request %d is %d bytes, over the %d the window allows", index+1, size, limit) + } + if index > 0 && !strings.Contains(messageText(ask[1]), "Summary so far:") { + t.Fatalf("request %d was not handed the summary so far", index+1) + } + } + messages := liveTranscript(agent) + if summaryNotes(messages) != 1 || !holdsText(messages, "question 10:") { + t.Fatal("the chunked summary did not land, or took the newest question") + } +} + +// THE REPORTED CASE, end to end: one question, one long answer, the next +// question refused for size. The free rungs protect all of it; the recovery's +// summary takes the first exchange and the refused request goes again. +func TestARefusedRequestIsRecoveredByASummary(t *testing.T) { + story := strings.Repeat("The watchmaker listened as the clocks kept time. ", 400) + completer := &scriptedCompleter{steps: []step{ + func(context.Context, []ai.Message) (*ai.Response, error) { + return nil, &provider.APIError{Status: 400, Overflow: true, Local: true, ContextLimit: 16_384, + InputTokens: 15_383, OutputTokens: 512, Message: "context needs shortening before sending"} + }, + func(_ context.Context, messages []ai.Message) (*ai.Response, error) { + if len(messages) != 2 || !strings.Contains(messageText(messages[1]), "PERSON: write me a story") { + t.Errorf("the second request was not the summary: %d messages", len(messages)) + } + return textResponse(summaryWords), nil + }, + func(_ context.Context, messages []ai.Message) (*ai.Response, error) { + if holdsText(messages, "The watchmaker listened") { + t.Error("the retried request still carried the story") + } + if !holdsText(messages, "now work in the streisand effect") { + t.Error("the retried request lost the question it was answering") + } + return textResponse("done"), nil + }, + }} + agent, _ := newTestAgent(t, completer, func(config *Config) { config.ContextWindow = 16_384 }) + agent.mu.Lock() + agent.messages = append(agent.messages, textMessage("user", "write me a story"), textMessage("assistant", story)) + agent.mu.Unlock() + for _, event := range collect(t, mustSubmit(t, agent, "now work in the streisand effect")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if completer.requests() != 3 { + t.Fatalf("requests = %d, want refusal, summary, retry", completer.requests()) + } +} + +// A SUMMARY IS DRAWN AS THE SESSION'S NOTE, never as words the person typed. +func TestASummaryIsDrawnAsANoteNotAPersonsMessage(t *testing.T) { + entries := shapeEntries([]ai.Message{ + textMessage("system", "prompt"), + textMessage("user", summaryNote(summaryWords, "grep or read x")), + textMessage("user", "the next question"), + }, nil) + if len(entries) != 2 || entries[0].Role != "note" || entries[1].Role != "user" { + t.Fatalf("entries = %+v", entries) + } +} + +// A SUMMARIZED CONVERSATION RESUMES AS ITSELF. The note rides the rebuilt +// window behind the compaction marker like any other line, so the reopened +// transcript is the one the pass left. +func TestASummarizedConversationResumesWithItsSummary(t *testing.T) { + path := filepath.Join(t.TempDir(), "session.jsonl") + live, workspace := newTestAgent(t, &summarizer{}, func(config *Config) { + config.ContextWindow = 65_536 + config.SessionFile = path + }) + personHeavy(live, 12, 20_000) + if _, err := live.compact(context.Background(), newEventHub()); err != nil { + t.Fatal(err) + } + want := liveTranscript(live) + if summaryNotes(want) != 1 { + t.Fatal("the live conversation was not summarized") + } + if err := live.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + resumed, err := newAgent(Config{ + Workspace: workspace, Model: "test/model", System: "SYSTEM", SessionFile: path, + }, &scriptedCompleter{}) + if err != nil { + t.Fatalf("reopen: %v", err) + } + t.Cleanup(func() { _ = resumed.Close() }) + got := liveTranscript(resumed) + if len(got) != len(want) || messageText(got[1]) != messageText(want[1]) { + t.Fatalf("resumed %d messages starting %.80q, want %d starting %.80q", + len(got), messageText(got[1]), len(want), messageText(want[1])) + } + for index := 2; index < len(want); index++ { + if messageText(got[index]) != messageText(want[index]) { + t.Fatalf("message %d differs after resume", index) + } + } +} + +// refusalOver is the local refusal of a request that is over the window by +// the given tokens, in the words the provider's budget guard uses. +func refusalOver(window, over int) *provider.APIError { + allowed := window - 512 - provider.ContextSafetyTokens(window) + return &provider.APIError{Status: 400, Overflow: true, Local: true, ContextLimit: window, + InputTokens: allowed + over, OutputTokens: 512, Message: "context needs shortening before sending"} +} + +// CODEAF'S OWN WORDS ARE NOT THE PERSON'S. The truncation continuation has no +// tag and reads as plain words; kept as the person's latest message, it pushed +// the request it continued into the summary (2026-09-28, "now translate to +// armenian"). +func TestTheContinuationNoteIsNotThePersonsLatestMessage(t *testing.T) { + agent, _ := newTestAgent(t, &summarizer{}, func(config *Config) { config.ContextWindow = 16_384 }) + agent.mu.Lock() + agent.messages = append(agent.messages, + textMessage("user", "tell me a story"), + textMessage("assistant", strings.Repeat("Elara walked the Whispering Woods. ", 120)), + textMessage("user", "now translate to armenian"), + textMessage("assistant", strings.Repeat("Էլարան քայլում էր անտառով։ ", 150)), + textMessage("user", truncationContinuationNote), + ) + persons := agent.personMessagesLocked() + agent.mu.Unlock() + if len(persons) != 2 || persons[1] != 3 { + t.Fatalf("person messages at %v, want [1 3]: the continuation was counted as the person's", persons) + } + if !agent.recoverContext(context.Background(), nil, refusalOver(16_384, 400)) { + t.Fatal("the refusal was not recovered") + } + messages := liveTranscript(agent) + if !holdsText(messages, "now translate to armenian") { + t.Fatal("the person's request was summarized away in favour of codeaf's continuation") + } +} + +// A REFUSAL A FEW TOKENS OVER IS RECOVERED, even when the only region left is +// smaller than an automatic pass would bother with — the earlier summary and +// one answer, as in the third refusal on 2026-09-28. +func TestARefusalJustOverTheWindowIsRecoveredFromASmallRegion(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 16_384 }) + agent.mu.Lock() + agent.messages = append(agent.messages, + textMessage("user", summaryNote(strings.Repeat("Elara protects the Whispering Woods. ", 50), "grep or read x")), + textMessage("user", truncationContinuationNote), + textMessage("assistant", strings.Repeat("Here is how the narrative could continue. ", 64)), + textMessage("user", "summarize the plot in 5 sentences please"), + ) + agent.mu.Unlock() + if !agent.recoverContext(context.Background(), nil, refusalOver(16_384, 78)) { + t.Fatal("a request 78 tokens over was left refused") + } + if model.calls() == 0 { + t.Fatal("no summary was written") + } + messages := liveTranscript(agent) + if summaryNotes(messages) != 1 || !holdsText(messages, "summarize the plot in 5 sentences please") { + t.Fatal("the recovery did not leave one note and the person's question") + } +} + +// THE REPLY BEFORE THE LATEST MESSAGE IS KEPT when that is enough: "translate +// it" is about the answer just given, and a summary cannot stand in for it. +func TestTheReplyTheLatestMessageIsAboutIsKept(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { config.ContextWindow = 65_536 }) + second := "SECOND STORY " + strings.Repeat("Elara found the lost library. ", 200) + agent.mu.Lock() + // The weight is the person's own notes, so the fold alone cannot free + // what is missing and the summary has to choose where to stop. + agent.messages = append(agent.messages, + textMessage("user", "tell me a story "+strings.Repeat("with dragons and a river and ", 800)), + textMessage("assistant", "Once upon a time."), + textMessage("user", "keep going "+strings.Repeat("and add a library and a key ", 800)), + textMessage("assistant", second), + textMessage("user", "translate it to spanish please"), + ) + agent.mu.Unlock() + if !agent.recoverContext(context.Background(), nil, refusalOver(65_536, 2_000)) { + t.Fatal("the refusal was not recovered") + } + messages := liveTranscript(agent) + if !holdsText(messages, "SECOND STORY") { + t.Fatal("the answer the person asked to translate was summarized away") + } + if holdsText(messages, "keep going") || summaryNotes(messages) != 1 { + t.Fatal("the recovery did not summarize the older notes") + } +} + +// A RECOVERY TAKES WHAT IS MISSING AND A LITTLE ROOM, not a quarter of the +// conversation. +func TestARecoveryReclaimsWhatIsMissingNotAQuarter(t *testing.T) { + agent, _ := newTestAgent(t, &refusingCompleter{t: t}, func(config *Config) { config.ContextWindow = 131_072 }) + agent.mu.Lock() + agent.messages = append(agent.messages, exchanges(40, map[int]string{})...) + for index := range agent.messages { + if agent.messages[index].Role == "assistant" { + agent.messages[index].Content = []ai.ContentPart{{Type: "text", Text: strings.Repeat("working through it. ", 200)}} + } + } + agent.mu.Unlock() + before := func() int { agent.mu.Lock(); defer agent.mu.Unlock(); return agent.transcriptTokensLocked() }() + if !agent.recoverContext(context.Background(), nil, refusalOver(131_072, 300)) { + t.Fatal("the refusal was not recovered") + } + after := func() int { agent.mu.Lock(); defer agent.mu.Unlock(); return agent.transcriptTokensLocked() }() + freed := before - after + if freed < 300 { + t.Fatalf("freed %d tokens, less than the 300 missing", freed) + } + if freed >= before/4 { + t.Fatalf("freed %d of %d tokens — a quarter, for 300 missing", freed, before) + } +} + +// /COMPACT SAYS WHY IT DID NOTHING, in terms of the conversation: how little +// there was since the last summary, or what stopped the summary. +func TestANoOpCompactSaysWhy(t *testing.T) { + short := func(agent *Agent, messages ...ai.Message) { + agent.mu.Lock() + agent.messages = append(agent.messages, messages...) + agent.mu.Unlock() + } + for _, test := range []struct { + name string + model Completer + messages []ai.Message + want string + }{ + {"only the latest message", &refusingCompleter{t: t}, + []ai.Message{textMessage("user", "hello")}, + "there is nothing before your latest message to summarize"}, + {"a little since the last summary", &refusingCompleter{t: t}, + []ai.Message{ + textMessage("user", summaryNote(strings.Repeat("the story so far. ", 150), "grep or read x")), + textMessage("user", "which book is longest?"), + textMessage("assistant", strings.Repeat("Order of the Phoenix. ", 60)), + textMessage("user", "and the first book?"), + }, + // Both messages since the summary are kept word for word, so + // nothing older is left for another one. + "nothing new since the last summary"}, + {"two messages and nothing older", &refusingCompleter{t: t}, + []ai.Message{ + textMessage("user", "question: "+strings.Repeat("pasted log line ", 800)), + textMessage("assistant", "answer"), + textMessage("user", "next question"), + }, + "there is nothing before your last 2 messages to summarize"}, + {"the summary failed", &summarizer{answer: func(int, []ai.Message) (*ai.Response, error) { + return nil, errors.New("provider unavailable") + }}, + []ai.Message{ + textMessage("user", "question: "+strings.Repeat("pasted log line ", 800)), + textMessage("assistant", "answer"), + textMessage("user", "second question"), + textMessage("assistant", "second answer"), + textMessage("user", "third question"), + textMessage("assistant", "third answer"), + textMessage("user", "fourth question"), + }, + "the model could not write a summary: provider unavailable"}, + } { + t.Run(test.name, func(t *testing.T) { + agent, _ := newTestAgent(t, test.model, func(config *Config) { config.ContextWindow = 131_072 }) + short(agent, test.messages...) + err := agent.Compact(context.Background()) + why, nothing := NothingToCompactWhy(err) + if !nothing || !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("Compact = %v, want nothing to compact", err) + } + if !strings.Contains(why, test.want) { + t.Fatalf("why = %q, want it to say %q", why, test.want) + } + }) + } +} + +// hintCount reads "<verb> N message(s)" out of a compaction hint, zero when +// the clause is absent. +func hintCount(hint, verb string) int { + var count int + for _, clause := range strings.Split(hint, " · ") { + if _, err := fmt.Sscanf(clause, verb+" %d message", &count); err == nil { + return count + } + } + return 0 +} + +// ONE /compact GOES ALL THE WAY. A conversation of pastes and long answers, +// already under the automatic target: the fold takes the answers, and the +// same pass goes on to summarize the pastes it cannot fold. It used to stop +// after the fold, because the fold alone had got under the automatic line, +// and a second /compact was the only way to the summary (sandbox #3, +// 2026-09-28: 1,083,488 → 567,975, then → 15,110). +func TestOneCompactFoldsAndThenSummarizesUnderTheAutomaticTarget(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(config *Config) { + config.ContextWindow = 131_072 + config.SessionFile = filepath.Join(t.TempDir(), "session.jsonl") + }) + agent.mu.Lock() + for turn := 1; turn <= 12; turn++ { + agent.messages = append(agent.messages, + textMessage("user", fmt.Sprintf("question %d: %s", turn, strings.Repeat("pasted log line ", 500))), + textMessage("assistant", fmt.Sprintf("answer %d: %s", turn, strings.Repeat("reading the parser. ", 400)))) + } + agent.mu.Unlock() + if before := estimate(agent); before >= agent.compactTargetTokens() { + t.Fatalf("fixture %d is not under the automatic target %d", before, agent.compactTargetTokens()) + } + + hub := newEventHub() + changed, err := agent.compactWithPolicy(context.Background(), hub, agent.requestedCompactPolicy()) + if err != nil || !changed { + t.Fatalf("compact = %v, %v; want one pass that changed the conversation", changed, err) + } + if model.calls() != 1 { + t.Fatalf("summary requests = %d, want the one pass to summarize", model.calls()) + } + event := lastCompacted(t, hub) + if hintCount(event.Hint, "folded") == 0 || hintCount(event.Hint, "summarized") == 0 { + t.Fatalf("hint = %q; want a fold and a summary in the same pass", event.Hint) + } + messages := liveTranscript(agent) + for turn := 10; turn <= 12; turn++ { + if !holdsText(messages, fmt.Sprintf("question %d:", turn)) { + t.Fatalf("recent question %d was summarized", turn) + } + } + if holdsText(messages, "question 1:") || summaryNotes(messages) != 1 { + t.Fatal("the older pastes were not summarized") + } + + // And a second /compact finds nothing left worth a request. + // It must not reach into the three kept messages to find something. + if err := agent.Compact(context.Background()); !errors.Is(err, ErrNothingToCompact) { + t.Fatalf("second /compact = %v, want nothing to compact", err) + } + if model.calls() != 1 { + t.Fatalf("summary requests = %d after a second /compact, want still 1", model.calls()) + } + if !holdsText(liveTranscript(agent), "question 10:") { + t.Fatal("a second /compact summarized the third-newest message") + } +} diff --git a/internal/session/compact_target_contract_test.go b/internal/session/compact_target_contract_test.go new file mode 100644 index 0000000000..ea21f0f9f1 --- /dev/null +++ b/internal/session/compact_target_contract_test.go @@ -0,0 +1,26 @@ +package session + +import ( + "context" + "strings" + "testing" +) + +func TestAutomaticPassSkipsSummaryWhenFixedPrefixCannotFit(t *testing.T) { + model := &summarizer{} + agent, _ := newTestAgent(t, model, func(c *Config) { + c.ContextWindow = 32_768 + c.System = strings.Repeat("fixed system instruction ", 3500) + }) + personHeavy(agent, 8, 4_000) + _, _ = agent.compact(context.Background(), newEventHub()) + if got := model.calls(); got != 0 { + t.Fatalf("automatic pass bought %d summaries for an unreachable target", got) + } + if !agent.recoverContext(context.Background(), nil, refusalOver(32_768, 2_000)) { + t.Fatal("refusal recovery did not shrink the conversation") + } + if got := model.calls(); got == 0 { + t.Fatal("refusal recovery skipped its useful summary") + } +} diff --git a/internal/session/compact_wait_contract_test.go b/internal/session/compact_wait_contract_test.go new file mode 100644 index 0000000000..874c921da0 --- /dev/null +++ b/internal/session/compact_wait_contract_test.go @@ -0,0 +1,139 @@ +package session + +import ( + "context" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/Agent-Field/agentfield/sdk/go/ai" +) + +// A refused turn waits for the pass already buying room, then sends the same +// question. The blocked model call makes the order observable without a clock. +func TestTurnWaitsForInFlightSummaryBeforeRecovering(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + refused := make(chan struct{}) + var requests atomic.Int32 + model := &summarizer{answer: func(_ int, messages []ai.Message) (*ai.Response, error) { + if strings.HasPrefix(messageText(messages[0]), "You are compacting") { + close(started) + <-release + return textResponse(summaryWords), nil + } + if requests.Add(1) == 1 { + close(refused) + return nil, refusalOver(16_384, 2_000) + } + return textResponse("ANSWER AFTER COMPACT"), nil + }} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 16_384 }) + personHeavy(agent, 10, 6_000) + compacted := make(chan error, 1) + go func() { compacted <- agent.Compact(context.Background()) }() + select { + case <-started: + case <-time.After(10 * time.Second): + t.Fatal("summary did not start") + } + events := make(chan []Event, 1) + go func() { events <- collect(t, mustSubmit(t, agent, "QUESTION DURING COMPACT")) }() + select { + case <-refused: + case <-time.After(10 * time.Second): + close(release) + t.Fatal("turn never reached its first request") + } + select { + case got := <-events: + close(release) + t.Fatalf("turn ended before the in-flight pass landed: %v", got) + case <-time.After(100 * time.Millisecond): + } + close(release) + if err := <-compacted; err != nil { + t.Fatal(err) + } + select { + case got := <-events: + for _, event := range got { + if event.Kind == EventError { + t.Fatalf("turn errored after the pass landed: %v", event.Err) + } + } + case <-time.After(10 * time.Second): + t.Fatal("turn did not finish after the pass landed") + } + if requests.Load() != 2 || model.calls() != 3 || !holdsText(liveTranscript(agent), "ANSWER AFTER COMPACT") { + t.Fatalf("requests=%d total calls=%d; want one summary and an answered retry", requests.Load(), model.calls()) + } +} + +func TestCancelledTurnStopsWaitingForInFlightSummary(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + refused := make(chan struct{}) + model := &summarizer{answer: func(_ int, messages []ai.Message) (*ai.Response, error) { + if strings.HasPrefix(messageText(messages[0]), "You are compacting") { + close(started) + <-release + return textResponse(summaryWords), nil + } + close(refused) + return nil, refusalOver(16_384, 2_000) + }} + agent, _ := newTestAgent(t, model, func(c *Config) { c.ContextWindow = 16_384 }) + personHeavy(agent, 10, 6_000) + compacted := make(chan error, 1) + go func() { compacted <- agent.Compact(context.Background()) }() + <-started + ctx, cancel := context.WithCancel(context.Background()) + events, err := agent.Submit(ctx, "CANCEL DURING COMPACT") + if err != nil { + close(release) + t.Fatal(err) + } + done := make(chan struct{}) + var kinds []EventKind + go func() { + for event := range events { + kinds = append(kinds, event.Kind) + if event.Kind == EventError { + t.Errorf("cancelled wait ended as an error: %v", event.Err) + } + } + close(done) + }() + <-refused + agent.mu.Lock() + waiting := agent.compacting + agent.mu.Unlock() + if !waiting { + t.Error("the turn was not refused while the pass was in flight") + } + select { + case <-done: + close(release) + t.Fatalf("turn ended before cancellation instead of waiting for the pass: %v", kinds) + case <-time.After(100 * time.Millisecond): + } + cancel() + select { + case <-done: + case <-time.After(10 * time.Second): + close(release) + t.Fatal("cancelled turn stayed behind the summary") + } + if len(kinds) == 0 || kinds[len(kinds)-1] != EventTurnDone { + t.Errorf("cancelled turn events = %v, want a stopped turn", kinds) + } + close(release) + if err := <-compacted; err != nil { + t.Fatal(err) + } +} + +// The screen joins scrollback above the live transcript's floor. A turn that +// finished during the summary must remain on that joined page after each pass. diff --git a/internal/session/compaction_refused_event_test.go b/internal/session/compaction_refused_event_test.go index fa7aab7803..c65033d0b3 100644 --- a/internal/session/compaction_refused_event_test.go +++ b/internal/session/compaction_refused_event_test.go @@ -1,20 +1,21 @@ package session -// A PASS THAT FOUND NOTHING SAYS SO ON THE EVENT ITSELF. +// A PASS THAT FOUND NOTHING SAYS NOTHING. // -// [EventCompacted] is sent on both paths by promise: a surface opens a row on -// [EventCompacting] and has to be able to settle it whether the pass edited -// anything or not. That left one value carrying two meanings, and the failing -// one was the silent one, so a reader could not tell a transcript that had been -// replaced from one that had not been touched. [Event.Unchanged] is the -// disjoint range, and its zero value is the meaning that was always safe. +// [EventCompacting] is sent only once a pass has really edited the transcript, +// and [EventCompacted] follows it at once. A pass that stubbed nothing and folded +// nothing therefore opens no row, and so has no row to settle: it returns +// [ErrNothingToCompact] to its caller and leaves the hub alone. An automatic +// attempt is silent that way, and `/compact` says "nothing to compact" from the +// error rather than from an event. // -// THIS PINS THE SESSION HALF ONLY. internal/tui3 pins what a surface does with -// the field, and that test passes on a hand-built event whether this half exists -// or not, which is exactly why both are written down. +// [Event.Unchanged] is still read by a surface, because a peer built before this +// announced every pass and settled a refused one with that field set +// (internal/tui3's compact_refused_test.go). No pass in this build sends it. import ( "context" + "errors" "strings" "testing" ) @@ -34,7 +35,7 @@ func lastCompacted(t *testing.T, hub *eventHub) Event { return Event{} } -func TestAPassThatCompactedNothingSaysTheTranscriptDidNotMove(t *testing.T) { +func TestAPassThatCompactedNothingSendsNoEvent(t *testing.T) { agent, _ := newTestAgent(t, &refusingCompleter{t: t}, func(config *Config) { config.ContextWindow = 2_000_000 }) @@ -43,17 +44,15 @@ func TestAPassThatCompactedNothingSaysTheTranscriptDidNotMove(t *testing.T) { agent.mu.Unlock() hub := newEventHub() - if _, err := agent.compact(context.Background(), hub); err != ErrNothingToCompact { + if _, err := agent.compact(context.Background(), hub); !errors.Is(err, ErrNothingToCompact) { t.Fatalf("compact = %v, want ErrNothingToCompact", err) } - event := lastCompacted(t, hub) - if !event.Unchanged { - t.Fatalf("a pass that stubbed nothing and folded nothing announced itself as a pass that happened: %+v", event) - } - // AND THE ROW STILL SETTLES. The field separates the two meanings; it does - // not withdraw the event, which a surface is waiting on either way. - if strings.TrimSpace(event.Hint) == "" { - t.Fatal("the refused pass settled the row with nothing to say") + hub.mu.Lock() + defer hub.mu.Unlock() + for _, event := range hub.backlog { + if event.Kind == EventCompacting || event.Kind == EventCompacted { + t.Fatalf("a pass that stubbed nothing and folded nothing announced itself: %+v", event) + } } } diff --git a/internal/session/compaction_test.go b/internal/session/compaction_test.go index 78918b91ad..546610c7ab 100644 --- a/internal/session/compaction_test.go +++ b/internal/session/compaction_test.go @@ -9,6 +9,7 @@ package session import ( "context" + "errors" "fmt" "os" "path/filepath" @@ -137,7 +138,7 @@ func TestCompactionWithNothingToDoSaysSo(t *testing.T) { agent.messages = append(agent.messages, exchanges(2, nil)...) agent.mu.Unlock() - if changed, err := agent.compact(context.Background(), nil); changed || err != ErrNothingToCompact { + if changed, err := agent.compact(context.Background(), nil); changed || !errors.Is(err, ErrNothingToCompact) { t.Fatalf("compact = %v, %v; want ErrNothingToCompact", changed, err) } } diff --git a/internal/session/connect.go b/internal/session/connect.go index 39d8a3581f..a900a1f1ab 100644 --- a/internal/session/connect.go +++ b/internal/session/connect.go @@ -590,6 +590,15 @@ func (a *Agent) armFamily(tools []bare.Tool) ([]string, error) { grownDefinitions = append(append(grownDefinitions, a.definitions...), definitions...) a.tools = grownTools a.definitions = grownDefinitions + for _, tool := range arriving { + known := false + for _, kept := range a.profileArmed { + known = known || kept.Name == tool.Name + } + if !known { + a.profileArmed = append(a.profileArmed, tool) + } + } return names, nil } diff --git a/internal/session/earlier_test.go b/internal/session/earlier_test.go index 5fccd5f378..218a60f49e 100644 --- a/internal/session/earlier_test.go +++ b/internal/session/earlier_test.go @@ -21,6 +21,7 @@ package session import ( "context" + "errors" "fmt" "path/filepath" "strings" @@ -326,7 +327,7 @@ func TestARefusedPassLeavesTheEarlierHistoryAlone(t *testing.T) { agent.messages = append(agent.messages, textMessage("user", "one short question")) agent.mu.Unlock() - if _, err := agent.compact(context.Background(), nil); err != ErrNothingToCompact { + if _, err := agent.compact(context.Background(), nil); !errors.Is(err, ErrNothingToCompact) { t.Fatalf("compact = %v, want ErrNothingToCompact", err) } if got := agent.EarlierHistory(); len(got.Entries) != 0 || got.Floor != 0 { diff --git a/internal/session/effortguard_test.go b/internal/session/effortguard_test.go index 8d6ca1f795..958634615d 100644 --- a/internal/session/effortguard_test.go +++ b/internal/session/effortguard_test.go @@ -43,15 +43,16 @@ var ladderStampers = map[string]string{ // site cannot join it by accident, because joining it means editing this map and // writing down which of the two things the call is. var legacyEffortStampers = map[string]string{ - "harness_build.go": "a design round's role call, from the designer tier's own suffix", - "harness_task.go": "carries a design's phase effort onto its node", - "task_shape.go": "the shaper's role call, from the shaper tier's own suffix", - "task_store.go": "rebuilds a design's phase effort off the checkpoint", - "task_restart.go": "restores the same saved design phase effort when retrying its checkpoint", - "orchestrate.go": "an adaptive-run node's role call, from the node's own tier", - "subharness_env.go": "a saved program's ai() call, from the options its author wrote", - "agent.go": "the per-model dial's doc comment, which names the shape it is NOT", - "callwindow.go": "an answer ask's thinking switched off: a requirement of that one call shape, not a person's depth", + "harness_build.go": "a design round's role call, from the designer tier's own suffix", + "harness_task.go": "carries a design's phase effort onto its node", + "task_shape.go": "the shaper's role call, from the shaper tier's own suffix", + "task_store.go": "rebuilds a design's phase effort off the checkpoint", + "task_restart.go": "restores the same saved design phase effort when retrying its checkpoint", + "orchestrate.go": "an adaptive-run node's role call, from the node's own tier", + "subharness_env.go": "a saved program's ai() call, from the options its author wrote", + "agent.go": "the per-model dial's doc comment, which names the shape it is NOT", + "callwindow.go": "an answer ask's thinking switched off: a requirement of that one call shape, not a person's depth", + "compact_summary.go": "a compaction summary asked at low effort: a digest of what already happened is that one call shape's own economy, not a person's depth", } // EVERY FILE THAT STAMPS A RUNG ALSO RESOLVES ONE. diff --git a/internal/session/loop.go b/internal/session/loop.go index eaeded4b8b..c21410364b 100644 --- a/internal/session/loop.go +++ b/internal/session/loop.go @@ -350,6 +350,13 @@ func (a *Agent) runTurn(ctx context.Context, hub *eventHub, user userMessage) bo // panic. defer a.endPhase() + if model := a.Model(); model != "" { + if err := a.preparePromptProfile(model); err != nil { + hub.send(Event{Kind: EventError, Err: err, Usage: a.sealTurn(turn, started, model)}) + return false + } + } + // BEFORE ANY OF IT: WHAT IS THIS SESSION WORKING TOWARDS? On an unattended // session with a budget the goal owner is a [Steward] (principal.go), and a // Steward that carries work on has to be carrying it on towards something. @@ -802,11 +809,9 @@ func (a *Agent) runTurn(ctx context.Context, hub *eventHub, user userMessage) bo // have been work is a question with no useful answer. usedTools := false - // overflowCompacted bounds the compact-and-retry answer to a context - // overflow at one pass per turn. A second overflow after a successful - // compaction is not a context problem this loop can fix by shrinking - // further, and retrying it forever would burn a summary call per attempt. - overflowCompacted := false + // Recovery is bounded per failed generation. A successful response resets + // the allowance: later work in the same turn may fill the window again. + overflowAttempts := 0 // A text-only answer normally closes the turn. A length stop is not an // answer, though: it is the provider saying that the answer did not fit, so @@ -1042,32 +1047,18 @@ func (a *Agent) runTurn(ctx context.Context, hub *eventHub, user userMessage) bo // CompactEnabled — that flag gates the automatic pass, not the // recovery from a request the provider has already refused. // - // AND IT IS THE VERDICT THAT SAYS SO, not a regex over the sentence. - // [taxonomy.ActionCompact] is the [taxonomy.Shape] policy's answer to - // a request that did not fit, and `overflowCompacted` is what it is - // told through [taxonomy.Evidence.Compacted] — so the once-per-turn - // rule is stated in the policy and read here rather than kept in two - // places that could come to disagree (taxonomy_boundary.go's - // [Agent.readOverflow]). - if a.readOverflow(err, model, overflowCompacted).Compacts() { - overflowCompacted = true - // AND THE REFUSAL IS THE ONE THING THAT TEACHES THE WINDOW. Every - // other figure in this law is a claim: the catalog's row, the - // surface's hint, this package's own default. A provider saying - // "that did not fit" is a measurement, and it is the only one - // available — so what was in front of it becomes the ceiling on - // this model's claim, here and in every later process - // ([Agent.learnServedWindow]). Until this line the loop compacted - // and re-sent and learned nothing, so the same over-long request - // was built again on the next long turn. - a.mu.Lock() - refused := a.estimateTokensLocked() - a.mu.Unlock() - a.learnServedWindow(model, refused) - if compacted, compactErr := a.compact(ctx, hub); compacted && compactErr == nil { + // The policy reads whether this generation exhausted its recovery + // allowance. Successful work below resets that allowance. + if a.readOverflow(err, model, overflowAttempts >= contextRecoveryAttempts).Compacts() { + overflowAttempts++ + if a.recoverContext(ctx, hub, err) { continue } } + if ctx.Err() != nil { + a.endStoppedTurn(ctx, hub, partial, turn, started, model) + return false + } // A permanent failure mid-stream is still a step the person // watched: the streamed text is kept and the turn is sealed, so an // error leaves the same record an interrupt does and the surface @@ -1077,6 +1068,7 @@ func (a *Agent) runTurn(ctx context.Context, hub *eventHub, user userMessage) bo return false } + overflowAttempts = 0 turn.Turns++ // WHO ANSWERED, AND WHAT THE RESCUE COST, folded onto what was timed // above. It is read here, before the money is banked, because it is the @@ -1998,6 +1990,9 @@ func (a *Agent) completeWithRetryReasoning(ctx context.Context, hub *eventHub, m purpose = purposeTask attemptCtx = provider.WithCallNode(attemptCtx, strconv.FormatUint(a.config.taskID, 10)) } + if err := a.preparePromptProfile(model); err != nil { + return nil, model, err + } messages, carried := a.snapshotWithReasoning() if wake, settle := settleWakeFrom(ctx); settle && wake.prompt != "" && len(messages) > 0 { rolePage := textMessage("system", strings.TrimSpace(wake.prompt)) @@ -2012,9 +2007,21 @@ func (a *Agent) completeWithRetryReasoning(ctx context.Context, hub *eventHub, m // The place a pointer may name is read ONCE for the whole request, under // the lock an anchor takes to move it (toolcompact.go). place := a.resultPlaceNow() - messages = a.compactToolHistory(messages, frozenToolHistory, + // The old boundary is an index into a transcript compaction may have + // rebuilt. The current floor moves with that rebuild and protects work + // from this turn from being mistaken for frozen history. + a.mu.Lock() + frozenThrough := min(frozenToolHistory, a.turnFloor) + a.mu.Unlock() + messages = a.compactToolHistory(messages, frozenThrough, func(message ai.Message) string { return a.fullResultPointer(message, place) }) attemptCtx = provider.WithMessageReasoning(attemptCtx, carried) + a.mu.Lock() + promptFloor := a.contextTokens + a.mu.Unlock() + attemptCtx = provider.WithContextBudget(attemptCtx, provider.ContextBudget{ + Window: a.window(), Reserve: ctxbudget.CompletionReserve(), PromptFloor: promptFloor, + }) attemptCtx, generation := a.beginGeneration(attemptCtx, reached) response, err := a.completeWithModel(attemptCtx, purpose, messages, model, ai.WithTools(a.beltDefinitions())) @@ -4175,6 +4182,7 @@ func (a *Agent) bank(call bankedCall) { } if call.context > 0 { a.contextTokens = call.context + a.contextBeltTokens = 0 } a.mu.Unlock() if call.ledger { @@ -4420,48 +4428,19 @@ const ( compactReservePercent = 15 compactReserveFloorTokens = 16384 - // compactKeepRecentTokens is how much of the tail survives verbatim. The - // summary is lossy by construction, so the recent work — the files just - // read, the error just seen — is kept as itself. + // compactKeepRecentTokens is the tail protected by routine cleanup. + // Manual and necessary reductions use compactRecentTokens instead. compactKeepRecentTokens = 20000 // bytesPerToken is the estimator used when no provider figure is - // available: ~4 bytes per token for code and English prose. It is only - // ever compared against a threshold with 16k of slack, so being 30% wrong - // moves when compaction fires, never whether the request fits. + // available: ~4 bytes per token for code and English prose. It remains an + // estimate; the final request check leaves margin and learns from refusals. bytesPerToken = 4 - // THE THRESHOLD FOLLOWS THE WINDOW, AND THE WINDOW IS THE MODEL CARD'S UNTIL - // AN ENDPOINT SAYS OTHERWISE. - // - // There used to be a flat ceiling here — twice [defaultContextWindow], so - // 256k — and it was put in for a real failure: the catalog row for - // ~deepseek/deepseek-v4-flash-latest claims 1,310,720 tokens, the trigger - // followed the claim to 1,114,112, a conversation grew to 386,309 tokens - // with compaction never once firing, and what came back at that size was the - // model's own template turned inside out. - // - // The ceiling answered that by disbelieving EVERY claim above 256k, and the - // bill for it was paid by every model that was telling the truth. Measured - // on 2026-08-31: a two-and-a-half-hour run on a model advertising 1.3M - // compacted nineteen times, each pass throwing away the prefix cache the run - // was otherwise getting 57–61% of its prompt back from, and the model was - // reduced to keeping its own notes file to survive the folding. - // - // So the ceiling is not a constant any more, it is a MEASUREMENT: the - // narrowest prompt this model has actually been refused for, learned from - // the overflow refusal itself and remembered across processes - // (internal/provider's NoteServedWindow, and the branch in [Agent.runTurn] - // that teaches it). A model nobody has refused is believed; one that has - // refused is capped at what it refused, for good. The 386k incident now - // costs one turn per model per machine instead of every model forever, and - // what it costs is paid by the model that earned it. - // - // Two guards stand behind that trade and neither is new. - // [Agent.guardOversizeRequest] still shrinks a transcript that has grown past - // the window before it goes out, and the reply guard cuts an answer that has - // stopped being language (internal/provider's CutBabble and CutMachinery), - // which is exactly the shape the 386k incident came back in. + // The catalog governs routine cleanup until this session learns an + // explicit endpoint window. Final request admission belongs to the provider, + // which also budgets schemas, replayed reasoning and the output allowance. + ) // window is the model's context in tokens, most specific answer first: the one @@ -4486,30 +4465,9 @@ func (a *Agent) trustedWindow() int { return trustedWindow(a.window(), int(a.servedWindow.Load())) } -// learnServedWindow is the closing half of the loop the threshold rides on: an -// endpoint has just refused a prompt for being too long, so what it refused is -// now the ceiling on that model's claim — in this session from the next check -// onward, and in every later process through the memo. -// -// It is called with the estimate that was refused rather than with a figure from -// the error, because no provider states one: what is known is that THIS many -// tokens was too many, here, and that is the honest ceiling. -func (a *Agent) learnServedWindow(model string, estimate int) { - if estimate <= 0 { - return - } - provider.NoteServedWindow(model, estimate) - if learned := provider.ServedWindow(model); learned > 0 { - a.servedWindow.Store(int64(learned)) - } -} - -// noteModelWindow refreshes what is known about the window of the model now in -// use. It is called wherever the model or the window moves, so that the memo a -// previous session wrote is in force from this session's first check. -func (a *Agent) noteModelWindow(model string) { - a.servedWindow.Store(int64(provider.ServedWindow(model))) -} +// noteModelWindow clears the active endpoint reading when the model changes. +// Durable endpoint evidence is applied by the provider at the request boundary. +func (a *Agent) noteModelWindow(_ string) { a.servedWindow.Store(0) } // childWindow is how large the window of the model a CHILD agent is about to run // on should be taken to be — a task node's worker, an adaptive run's worker, a @@ -4552,30 +4510,13 @@ func (a *Agent) compactThreshold() int { return compactThresholdOf(a.trustedWindow()) } -// TrustedWindow is a claimed context window with everything this process has -// learned applied over it — the figure the compaction machinery works from, as -// opposed to [Agent.window], which stays the model's own claim because the -// status meter is describing the model rather than this law. -// -// Called without a model there is nothing to have learned, so the claim comes -// back untouched; [TrustedWindowFor] is the door that applies a model's memo. -// -// It is exported for the same reason [CompactThreshold] is: a surface that -// needs to know how much room the guard leaves must read the guard, not a -// second copy of it. -func TrustedWindow(window int) int { return TrustedWindowFor("", window) } - -// TrustedWindowFor is [TrustedWindow] for a NAMED model: the claim, capped by -// the narrowest prompt that model has been refused for, when this process has -// ever seen it refused. -// -// An empty model, or one nothing has been learned about, gets its claim back -// unchanged — which is the ordinary case and the emptiness law applied to a -// measurement: "nothing was learned" may not be spelled the same way as "this -// model has no room". -func TrustedWindowFor(model string, window int) int { - return trustedWindow(window, provider.ServedWindow(model)) -} +// TrustedWindow returns a catalog window without guessing an endpoint. +// The provider applies endpoint-specific evidence to each assembled request. +func TrustedWindow(window int) int { return window } + +// TrustedWindowFor preserves the surface API, but a model-only memo is no +// longer a context limit. Old served_window entries were rejected prompt sizes. +func TrustedWindowFor(_ string, window int) int { return window } // trustedWindow is the law itself over two plain numbers, so that an agent // holding a learned figure of its own and a surface asking about a model by name @@ -4764,7 +4705,7 @@ func (a *Agent) compactTargetTokens() int { // keepRecentTokens is the verbatim tail budget, capped at a quarter of the // window. Keeping 20k of a 200k window is a tail; keeping 20k of an 8k window // is not a compaction at all, and without the cap a small-window session would -// find nothing to summarize and overflow with the pass "succeeding". +// find nothing to fold and overflow with the pass "succeeding". func (a *Agent) keepRecentTokens() int { return keepRecent(a.trustedWindow()) } @@ -4794,8 +4735,9 @@ func keepRecent(window int) int { // costs the turn: 386,309 tokens went out against a row claiming 1.3M and came // back as corrupted template text rather than an error anything could catch. // -// The bar is [TrustedWindow], not the model's own claim, for exactly that -// reason — the claim is what was wrong. +// This early estimate excludes encoded schemas and framing. The provider's +// final encoding guard checks those, output allowance and the serving limit +// together before any request can leave. func (a *Agent) guardOversizeRequest(ctx context.Context, hub *eventHub) { ceiling := a.trustedWindow() if ceiling <= 0 { @@ -4807,9 +4749,9 @@ func (a *Agent) guardOversizeRequest(ctx context.Context, hub *eventHub) { if estimate <= ceiling { return } - // A pass that finds nothing is not an error here: the transcript is then - // the person's own words and the recent tail, and the step goes out because - // there is nothing left to take out of it. + // A no-op here still reaches the provider's final encoding guard. If the + // protected content cannot fit, that guard returns a local context refusal + // for the bounded emergency pass instead of sending it upstream. _, _ = a.compact(ctx, hub) } @@ -4831,39 +4773,99 @@ func (a *Agent) maybeCompact(ctx context.Context, hub *eventHub) { _, _ = a.compact(ctx, hub) } -// ErrNothingToCompact says a compaction pass had nothing to do: the whole -// transcript already fits inside the keep-recent tail. It is a sentinel rather -// than a silent no-op so a surface's /compact can say "nothing to compact" +// ErrNothingToCompact says a pass found no eligible history to reduce. +// User instructions and recent work may still fill the window. It is a +// sentinel rather than a silent no-op so a surface's /compact can say "nothing to compact" // instead of reporting a success that changed nothing. var ErrNothingToCompact = errors.New("session: nothing to compact") +// NothingToCompact is [ErrNothingToCompact] with the reason a person can act +// on: how little older conversation there was, or why the summary that would +// have shortened it did not land. errors.Is still matches the sentinel. +type NothingToCompact struct{ Why string } + +func (e *NothingToCompact) Error() string { + if e.Why == "" { + return ErrNothingToCompact.Error() + } + return ErrNothingToCompact.Error() + ": " + e.Why +} + +func (e *NothingToCompact) Is(target error) bool { return target == ErrNothingToCompact } + +// NothingToCompactWhy reads the reason out of a no-op, and reports false for +// any other error. It reads the words as well as the type, because a remote +// engine's error reaches a surface as its text alone. +func NothingToCompactWhy(err error) (string, bool) { + if err == nil { + return "", false + } + var typed *NothingToCompact + if errors.As(err, &typed) { + return typed.Why, true + } + text := err.Error() + if !strings.HasPrefix(text, ErrNothingToCompact.Error()) { + return "", false + } + return strings.TrimSpace(strings.TrimPrefix(strings.TrimPrefix(text, ErrNothingToCompact.Error()), ":")), true +} + +// SummarySkipped says a pass shortened the conversation while its summary did +// not land. It travels as a result of /compact so the surface can report both +// facts even when the pass ran without an event hub. +type SummarySkipped struct{ Why string } + +func (e *SummarySkipped) Error() string { return "session: compacted: summary skipped: " + e.Why } + +// SummarySkippedWhy also reads remote errors that carry only their text. +func SummarySkippedWhy(err error) (string, bool) { + if err == nil { + return "", false + } + var typed *SummarySkipped + if errors.As(err, &typed) { + return typed.Why, true + } + const prefix = "session: compacted: summary skipped: " + if strings.HasPrefix(err.Error(), prefix) { + return strings.TrimPrefix(err.Error(), prefix), true + } + return "", false +} + // ErrCompactionInFlight says another pass is already running. The second caller // gets an error for the same reason: it did nothing, and it should say so. var ErrCompactionInFlight = errors.New("session: a compaction pass is already running") // compactionPass is what one pass did, and it is the ONLY thing a pass produces: -// two counts and, when something was folded, the marker line that stands in its -// place. There is no summary because there is no summarizer — a pass is a -// rearrangement of text this session already has (see the file header comment on -// [Agent.compact]). +// its counts and, when something was folded, the marker line that stands in its +// place. A summary, when the last rung writes one, is in the transcript like any +// message; the pass keeps only how many messages it replaced. type compactionPass struct { stubbed int folded int - marker string + // summarized is how many messages the pass's summary note replaced + // (compact_summary.go), zero when the free rungs were enough. + summarized int + marker string + // summarySkipped explains an attempted summary that did not land after + // the free rungs changed the conversation. + summarySkipped string // stored says the full record went somewhere a later session can still read // it — the store's thread (chatlog.go). It is what makes the difference // between the two announce lines honest. stored bool } -func (p compactionPass) empty() bool { return p.stubbed == 0 && p.folded == 0 } +func (p compactionPass) empty() bool { return p.stubbed == 0 && p.folded == 0 && p.summarized == 0 } -// compact runs one pass, and IT MAKES NO MODEL CALL AT ALL. +// compact runs one pass, and IT MAKES NO MODEL CALL UNLESS THE FREE RUNGS FAIL. // -// The old pass paid a summarizer to write prose about the prefix it was about to -// throw away. It was expensive at the worst moment, it was lossy by +// The pass before 2026-08-18 paid a summarizer to write prose about the prefix +// on every compaction. It was expensive at the worst moment, it was lossy by // construction, and the loss was unrecoverable because the transcript the prose -// was written from went with it. What replaces it is two mechanical passes over +// was written from went with it. What replaced it is two mechanical passes over // the same messages, in order of how cheap the content is to give up: // // 1. THE STUB PASS. A tool result the model has already used is a pointer to @@ -4882,15 +4884,31 @@ func (p compactionPass) empty() bool { return p.stubbed == 0 && p.folded == 0 } // What the model is handed instead of a summary is the STATE CARD, which rides // in the system prompt on every turn and is maintained incrementally by the // post-turn extractor (card.go). So the cost of knowing what the conversation is -// about is amortized across the turns that produced it, and the compaction -// itself is free. -// -// The lock is held across the WHOLE pass, which the old one could not do because -// it was waiting on a provider. That is not a cost, it is the removal of one: -// the mid-batch race the old pass had to repair — a tool result landing after -// the cut while the summary was being written — cannot happen when nothing is -// awaited. -func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { +// about is amortized across the turns that produced it, and those two rungs are +// free. +// +// 3. THE SUMMARY, and only when the two above leave the conversation over the +// line the pass needs (compact_summary.go). They never touch a person's +// words or the turn in hand, so a conversation made mostly of those had no +// way down at all: every pass found nothing while the window filled. The +// conversation's own model rewrites the oldest part — person and assistant +// alike — as one note, keeping the most recent person messages whole and +// the original lines in the journal. +// +// The lock is held across the two free rungs. The summary is the one wait on a +// provider, and it is made with the lock released and the region compared +// before the answer is spliced in; the old summarizer held the lock and could +// not be interrupted. +func (a *Agent) compact(ctx context.Context, hub *eventHub) (bool, error) { + return a.compactWithPolicy(ctx, hub, a.automaticCompactPolicy()) +} + +func (a *Agent) compactWithPolicy(ctx context.Context, hub *eventHub, policy compactPolicy) (bool, error) { + changed, _, err := a.compactWithPolicyResult(ctx, hub, policy) + return changed, err +} + +func (a *Agent) compactWithPolicyResult(ctx context.Context, hub *eventHub, policy compactPolicy) (bool, string, error) { // A PASS SAYS ITSELF WHILE IT RUNS, and it says itself from OUTSIDE the // lock. A phase post reaches a surface, and a surface answers one by asking // for a frame — so a phase posted with this agent's mutex held is a surface @@ -4899,23 +4917,19 @@ func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { // is refused below still ends this clock on the way out. a.tellPhase(provider.PhaseTidying, "the conversation", time.Now()) defer a.endPhase() + // The definitions are outside the transcript and take the belt's own lock. + // Read their weight before taking the session lock, then use it only for + // the person's meter; the policy's transcript-only lines stay unchanged. + belt := a.beltTokens() a.mu.Lock() if a.compacting { a.mu.Unlock() - return false, ErrCompactionInFlight + return false, "", ErrCompactionInFlight } a.compacting = true - tokensBefore := a.estimateTokensLocked() - // AND THE PASS IS ANNOUNCED THE MOMENT IT BEGINS, not only when it ends. - // [EventCompacting] has said in its own doc comment since it was declared - // that a pass is visible while it runs, and until this line nothing in the - // repository ever sent it — three handlers in internal/tui3 waited on an - // event with no sender, so a person watching a turn stop to tidy itself saw - // the finished line and never the work. The event goes out under the lock - // deliberately: [eventHub.send] only appends to queues and cannot block, - // which is what makes it safe here and a phase post not. - hub.send(Event{Kind: EventCompacting, Hint: "compacting " + approxTokens(tokensBefore) + " tokens"}) + a.compactDone = make(chan struct{}) + tokensBefore := max(a.estimateTokensLocked(), a.transcriptTokensLocked()+belt) // THE CONVERSATION IS SHAPED FOR THE SCROLLBACK BEFORE IT IS EDITED. This is // the same region a resume recovers from the journal ([replayedSession.earlier]), @@ -4936,7 +4950,9 @@ func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { earlier := shapeEntries(a.messages, a.file, a.presentation) pass := compactionPass{stored: a.chatlog != nil} - pass.stubbed = a.stubOldOutputsLocked() + if !policy.active { + pass.stubbed = a.stubOldOutputsLocked() + } // THE FOLD IS ASKED FOR AGAINST THE TARGET, NOT THE TRIGGER. The pass fires // at the threshold, and the stub pass alone routinely lands the estimate just // under it — below the trigger, above the target, with no headroom at all. The @@ -4944,28 +4960,58 @@ func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { // thousand tokens crossed the line again, and the session was back in exactly // the once-per-step thrash [compactTarget] exists to end. What buys the // headroom is the same figure the fold already stops at. - if a.estimateTokensLocked() > a.compactTargetTokens() { - pass.folded, pass.marker = a.foldLocked() + if policy.active || a.estimateTokensLocked() > policy.target { + pass.folded, pass.marker = a.foldWithPolicyLocked(policy) + } + // THE LAST RUNG IS A SUMMARY, and it is the only one that waits on a model + // (compact_summary.go). The lock is released for the call and taken back + // to splice the answer in; [Agent.compacting] still holds every other pass + // off, and a transcript that moved in the gap keeps its shape. + // why is what a pass that changed nothing says about itself (a person's + // /compact reads it; an automatic pass says nothing either way). + why := "" + appendedDuringSummary := 0 + attempted := false + if summaryWanted(policy, pass, a.transcriptTokensLocked()) { + plan, short, ok := a.planSummaryLocked(policy) + why = short + if ok { + attempted = true + lengthBeforeSummary := len(a.messages) + a.mu.Unlock() + a.tellPhase(provider.PhaseTidying, "summarizing the conversation", time.Now()) + summary, err := a.writeSummary(ctx, plan) + a.mu.Lock() + appendedDuringSummary = max(0, len(a.messages)-lengthBeforeSummary) + if err == nil { + pass.summarized, why = a.spliceSummaryLocked(plan, summary) + } else { + why = summaryFailedWhy(ctx, err) + } + } + } + // ONLY A SUMMARY THAT WAS ASKED FOR AND DID NOT LAND IS "SKIPPED". Too + // little older conversation to be worth a call is not a failure: a fold + // that changed something says what it folded, as it always did, and the + // reason is kept for the pass that changed nothing ([NothingToCompact]). + if attempted && pass.summarized == 0 { + pass.summarySkipped = why } a.compacting = false + close(a.compactDone) + a.compactDone = nil if pass.empty() { - // Nothing was old enough to stub and nothing was foldable: the whole - // transcript is the person's own words and the recent tail, which is - // what [ErrNothingToCompact] has always meant. + // No eligible material was reduced. A manual pass and a routine pass + // protect different tails, so this is not a claim about total size. a.mu.Unlock() - // AND [EventCompacted] FOLLOWS [EventCompacting] ON EVERY PATH, which is - // the promise the pair is declared with (session.go). A surface opens a - // row on the first and settles it on the second; a pass that announced - // itself and then said nothing would leave that row open for the rest of - // the session, so a pass that found nothing says exactly that. - // AND IT SAYS WHICH OF THE TWO THINGS THIS EVENT MEANS. The kind alone - // cannot: it is sent on both paths, so a reader that rebased on it - // rebased on a replacement that did not happen ([Event.Unchanged]). - hub.send(Event{Kind: EventCompacted, Hint: "nothing to compact", Unchanged: true}) - return false, ErrNothingToCompact + return false, "", &NothingToCompact{Why: why} } + // Mechanical reduction is synchronous. Publish a paired seam only once + // it changed something; an automatic no-op leaves no transcript row. + hub.send(Event{Kind: EventCompacting, Hint: "compacting " + approxTokens(tokensBefore) + " tokens"}) + // The pass really edited the transcript, so the region above it is now // history and this is the record of it. The region a PREVIOUS pass left is // replaced rather than prepended to, which is the same one-hop reading the @@ -4985,12 +5031,15 @@ func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { // time, under this lock, at the one moment a person is most likely to be // pressing Esc ([Agent.Interrupt] wants the same lock). a.earlier = earlier - a.earlierFloor = countEntries(a.messages) + a.earlierFloor = countEntries(a.messages[:len(a.messages)-appendedDuringSummary]) // The provider's context figure described the request that is now gone. - // Zero sends the estimator back to the content until the next response. + // Estimate the rebuilt transcript AND the definitions the next request + // carries. Keep that weight as the transcript grows until a provider's + // next reported count replaces it. a.contextTokens = 0 - tokensAfter := a.estimateTokensLocked() + a.contextBeltTokens = belt + tokensAfter := a.meterTokensLocked() // The whole rebuilt window is re-journaled behind the marker, not just the // tail: a stub and a fold are edits to messages the file already holds ABOVE // the marker, and replay discards everything above it. Writing the window is @@ -5005,9 +5054,9 @@ func (a *Agent) compact(_ context.Context, hub *eventHub) (bool, error) { a.mu.Unlock() if hub != nil { - hub.send(Event{Kind: EventCompacted, Hint: compactionHint(pass, tokensBefore, tokensAfter)}) + hub.send(Event{Kind: EventCompacted, Hint: compactionHint(pass, tokensBefore, tokensAfter), Summarized: pass.summarized}) } - return true, nil + return true, pass.summarySkipped, nil } // compactionHint is the one dim line the turn after a pass shows, and every @@ -5028,7 +5077,13 @@ func compactionHint(pass compactionPass, before, after int) string { if pass.folded > 0 { clauses = append(clauses, fmt.Sprintf("folded %d message%s", pass.folded, plural(pass.folded))) } - if before > after { + if pass.summarized > 0 { + clauses = append(clauses, fmt.Sprintf("summarized %d message%s", pass.summarized, plural(pass.summarized))) + } + if pass.summarySkipped != "" { + clauses = append(clauses, "summary skipped: "+pass.summarySkipped) + } + if before > after && approxTokens(before) != approxTokens(after) { clauses = append(clauses, fmt.Sprintf("%s → %s tokens", approxTokens(before), approxTokens(after))) } if pass.stored { @@ -5066,15 +5121,30 @@ func compactionHint(pass compactionPass, before, after int) string { // foldable material before it gets there still succeeds with what it took — // the target is how far to go, never a condition on the pass. func (a *Agent) foldLocked() (int, string) { + return a.foldWithPolicyLocked(compactPolicy{target: a.compactTargetTokens(), keep: a.keepRecentTokens()}) +} + +func (a *Agent) foldWithPolicyLocked(policy compactPolicy) (int, string) { a.alignReasoningLocked() - limit := a.cutPointLocked() - protectTurn := a.turnContinuesLocked() - target := a.compactTargetTokens() * bytesPerToken + limit := a.cutPointForLocked(policy.keep) + if policy.active { + // The newest call and all of its answers remain whole even when one + // batch alone exceeds the recent allowance. Pending calls never fold. + for i := len(a.messages) - 1; i > 0; i-- { + if a.messages[i].Role == "assistant" { + limit = min(limit, i) + break + } + } + } + protectTurn := !policy.active && a.turnContinuesLocked() + target := policy.target * bytesPerToken total := 0 - for _, message := range a.messages { - total += messageBytes(message) + for i := range a.messages { + total += a.transcriptMessageBytesLocked(i) } + beforeTotal := total folded := make(map[int]bool, 16) first, last := -1, -1 for index := 1; index < limit && total > target; { @@ -5102,7 +5172,7 @@ func (a *Agent) foldLocked() (int, string) { // A stub is useful only while its tool call remains in the window. Keep // tool batches intact: folding the assistant call would either orphan the // stub or fold the stub too, defeating the required stubs-plus-folds shape. - if len(a.messages[index].ToolCalls) > 0 { + if !policy.active && len(a.messages[index].ToolCalls) > 0 { keepsStub := false for cursor := index + 1; cursor < batch; cursor++ { if strings.HasPrefix(strings.TrimSpace(messageContentText(a.messages[cursor])), stubMarker) { @@ -5117,7 +5187,7 @@ func (a *Agent) foldLocked() (int, string) { } for cursor := index; cursor < batch; cursor++ { folded[cursor] = true - total -= messageBytes(a.messages[cursor]) + total -= a.transcriptMessageBytesLocked(cursor) if first < 0 { first = cursor } @@ -5137,6 +5207,9 @@ func (a *Agent) foldLocked() (int, string) { // when this run actually reached it, since a post can fail and a fold that // sends the model to a store holding nothing is the dead pointer again. marker := foldMarker(len(folded), journal, from, to, a.chatlog.ref(a.messages[first]) != "") + if total+messageBytes(textMessage("user", marker)) >= beforeTotal { + return 0, "" + } rebuilt := make([]ai.Message, 0, len(a.messages)-len(folded)+1) rebuiltReasoning := make([]provider.MessageReasoning, 0, cap(rebuilt)) rebuilt = append(rebuilt, a.messages[0]) @@ -5249,12 +5322,14 @@ func foldLineSpan(from, to int) string { // cutPointLocked walks back from the tail until the keep-recent budget is // spent and returns the index the kept tail starts at. -func (a *Agent) cutPointLocked() int { - budget := a.keepRecentTokens() * bytesPerToken +func (a *Agent) cutPointLocked() int { return a.cutPointForLocked(a.keepRecentTokens()) } + +func (a *Agent) cutPointForLocked(keep int) int { + budget := keep * bytesPerToken cut := len(a.messages) // index 0 is the system message; it is never summarized and never cut. for cut > 1 { - size := messageBytes(a.messages[cut-1]) + size := a.transcriptMessageBytesLocked(cut - 1) if budget-size < 0 { break } @@ -5278,11 +5353,22 @@ func (a *Agent) cutPointLocked() int { // threshold. Taking the max keeps the honest number as a floor while letting // the content speak for everything after it. func (a *Agent) estimateTokensLocked() int { - total := 0 - for _, message := range a.messages { - total += messageBytes(message) + estimate := a.transcriptTokensLocked() + if a.contextTokens > estimate { + return a.contextTokens } - estimate := EstimateTokens(total) + return estimate +} + +// meterTokensLocked is what a person is shown: [Agent.estimateTokensLocked] +// plus, between a compaction and the provider's next count, the tool +// definitions the next request carries. THE THRESHOLDS DO NOT READ IT. The fold +// and the automatic trigger are measured on the transcript estimate their laws +// were written against (compaction_headroom_test.go); only the status line, the +// /compact note and the ⚭ figures need the whole request, so that a pass does not +// read as a larger drop than the next request will show. +func (a *Agent) meterTokensLocked() int { + estimate := a.transcriptTokensLocked() + a.contextBeltTokens if a.contextTokens > estimate { return a.contextTokens } diff --git a/internal/session/memory.go b/internal/session/memory.go index 2e5fd67e55..995148779f 100644 --- a/internal/session/memory.go +++ b/internal/session/memory.go @@ -128,10 +128,12 @@ func newMemoryBrain(s *store.Store) *memoryBrain { return &memoryBrain{store: s, reflex: &reflex.Session{}} } -// remembers reports whether this session has a brain at all. Every entry point -// in this file asks it first, and the answer is a fact about the wiring rather -// than about a setting: the door opens no store when memory is off. -func (a *Agent) remembers() bool { return a.memory != nil && a.memory.store != nil } +// remembers gates every memory entry point. An automatic lean conversation +// may hold a dormant brain for a later model switch, but performs no memory +// work until its profile is full. +func (a *Agent) remembers() bool { + return a.memory != nil && a.memory.store != nil && !a.config.promptProfile().lean() +} // memorySourceSession is the journal header id attached to a memory write. A // test or embedded session without a journal still has the stable session name diff --git a/internal/session/prompt.go b/internal/session/prompt.go index 6739bc99d3..18e5d2c148 100644 --- a/internal/session/prompt.go +++ b/internal/session/prompt.go @@ -309,6 +309,10 @@ func renderSystemAt(config Config, now time.Time) string { out.WriteString(workerFooter(config, now)) return out.String() } + // A MODEL WITH NO TOOLS READS A PAGE WITHOUT THEM (chatpage.go). + if config.promptProfile().chat() { + return chatPage(config, now) + } var out strings.Builder // THE PAGE, WITH ITS TOOL-NAMING FACTS COMPOSED FROM THIS BELT'S OWN // PREDICATES (beltfacts.go). Everything below conditions a whole page on diff --git a/internal/session/promptprofile.go b/internal/session/promptprofile.go index 4f463b6977..350afd7350 100644 --- a/internal/session/promptprofile.go +++ b/internal/session/promptprofile.go @@ -97,6 +97,7 @@ package session import ( "strings" + "sync/atomic" "github.com/Agent-Field/codeaf/internal/env" @@ -116,10 +117,21 @@ const ( profileFull promptProfile = config.PromptProfileFull // profileLean is the Pi-sized prefix. profileLean promptProfile = config.PromptProfileLean + // profileChat is the prefix of a model that takes no tools (chatpage.go). + // It is not a word a person types: it is decided by what the model can + // do, above the pin, the row and the window, and it ends when the + // conversation moves to a model that can use tools. + profileChat promptProfile = "chat" ) -// lean is the one question the rest of the package asks of a profile. -func (p promptProfile) lean() bool { return p == profileLean } +// lean is the one question the rest of the package asks of a profile. The +// chat prefix answers yes: everything a lean prefix gives up — memory, the +// second instruction file, the shelf — a conversation with no tools gives up +// too. +func (p promptProfile) lean() bool { return p == profileLean || p == profileChat } + +// chat says the prefix is the tool-less page with an empty belt. +func (p promptProfile) chat() bool { return p == profileChat } const ( // leanWindowThreshold is the window under which the prefix goes lean, in @@ -153,19 +165,24 @@ const ( // ── the decision ──────────────────────────────────────────────────────────── -// promptProfile is this session's profile, SETTLED ONCE at construction -// (agent.go's newAgent) and read from the config everywhere after. -// -// It is a method on [Config] and not on [Agent] because every reader of it is a -// reader the agent does not exist for yet: the page is rendered before the agent -// is built, and the belt is built from the same config a moment later. That is -// the same law beltfacts.go's predicates are written under — every predicate is -// answerable from the config alone — and it is what makes it impossible for the -// page and the belt to disagree about which profile this is. -// -// A config nobody settled derives the answer live, which is what a test asking -// the question of a bare [Config] wants. +// livePromptProfile keeps the launch choice and the current automatic shape. +// The pointer is private to one engine, even when its Config came from a parent. +// Background memory readers share the atomic value; request boundaries own +// changes to the page and belt. +type livePromptProfile struct { + current atomic.Value + auto bool + // launch is the profile an explicit choice named, which a conversation + // returns to when it leaves a model that takes no tools. Unused when auto. + launch promptProfile +} + +// promptProfile also works before an engine exists, when the page and belt +// are first composed from a bare Config. func (c Config) promptProfile() promptProfile { + if c.liveProfile != nil { + return c.liveProfile.current.Load().(promptProfile) + } if c.profile != "" { return c.profile } @@ -189,6 +206,20 @@ func settlePromptProfile(c Config) promptProfile { return resolvePromptProfile(c // changes it. Both answer `auto` by saying nothing, and then the window decides // exactly as it did before either existed. func resolvePromptProfile(c Config) promptProfile { + // A MODEL THAT TAKES NO TOOLS IS ABOVE EVERY RUNG. The pin, the row and the + // window choose how much of the working page to send; none of them can + // make a model use a tool, and a page about tools it cannot call is a page + // that lies (chatpage.go). + if c.takesNoTools(c.Model) { + return profileChat + } + return chosenPromptProfile(c) +} + +// chosenPromptProfile is the ladder under the capability: the pin, the row, +// then the window. A live conversation returns to it when it moves from a +// model with no tools to one that has them. +func chosenPromptProfile(c Config) promptProfile { if pinned, ok := promptProfileWord(env.Get(promptProfileEnv)); ok { return pinned } @@ -513,3 +544,14 @@ func (c Config) instructionLimit() int { // the second copy is the first thing to go — and it is the second FOUND rather // than a named file, so a project that has only CLAUDE.md still gets its rules. func (c Config) onlyOneInstructionFile() bool { return c.promptProfile().lean() } + +// takesNoTools says the catalog knows this model and says it takes no tool +// calls — the same test internal/provider's toolless.go leaves the tools off +// by. Unknown is false: a model nobody has described keeps the working page. +func (c Config) takesNoTools(model string) bool { + if c.SupportsParameter == nil { + return false + } + supported, known := c.SupportsParameter(model, "tools") + return known && !supported +} diff --git a/internal/session/promptprofile_live.go b/internal/session/promptprofile_live.go new file mode 100644 index 0000000000..2fa7fb2cf2 --- /dev/null +++ b/internal/session/promptprofile_live.go @@ -0,0 +1,103 @@ +package session + +import ( + "fmt" + "time" + + "github.com/Agent-Field/codeaf/internal/exec/bare" +) + +// preparePromptProfile changes a conversation's prefix at a request boundary. +// Nothing already in flight is rewritten. Caller prompts and deliberately +// narrowed workers keep the shape their owner chose; an explicit lean or full +// choice keeps its word, and gives way only to a model that cannot use tools +// (chatpage.go), coming back when the conversation leaves that model. +func (a *Agent) preparePromptProfile(model string) error { + a.mu.Lock() + defer a.mu.Unlock() + state := a.config.liveProfile + if state == nil || !a.systemOwn || a.config.InTask { + return nil + } + next := state.launch + if state.auto { + window := a.window() + if a.config.ContextWindowFor != nil { + if known := a.config.ContextWindowFor(model); known > 0 { + window = known + } + } + next = profileFull + if window < leanWindowThreshold { + next = profileLean + } + } + if a.config.takesNoTools(model) { + next = profileChat + } + previous := a.config.promptProfile() + if previous == next { + return nil + } + a.armMu.Lock() + if a.withdrawn != nil { + a.armMu.Unlock() + return nil + } + oldShelf, oldOrder, oldPrearm := a.shelf, a.shelfOrder, a.prearm + a.armMu.Unlock() + + // The pointer is stable for this engine's lifetime; background memory + // readers can observe its value without racing a write to the Config. + state.current.Store(next) + tools := a.belt() + a.armMu.Lock() + // Explicitly loaded capabilities and connected services remain usable. + // Only the default belt is repartitioned when the window changes. + held := make(map[string]bool, len(tools)) + for _, tool := range tools { + held[tool.Name] = true + } + // A MODEL WITH NO TOOLS TAKES NONE BACK. What was loaded stays in + // profileArmed and returns with the next model that can use it. + if next.chat() { + tools = nil + } + for _, tool := range a.profileArmed { + if next.chat() { + break + } + if !held[tool.Name] { + tools = append(tools, tool) + held[tool.Name] = true + } + } + // Pre-armed groups join this same atomic publication, so no request can + // see the new page without the tools that page promises. + for _, tool := range a.prearm { + if !held[tool.Name] { + tools = append(tools, tool) + held[tool.Name] = true + } + } + definitions, err := toolDefinitions(tools) + if err != nil { + a.shelf, a.shelfOrder, a.prearm = oldShelf, oldOrder, oldPrearm + state.current.Store(previous) + a.armMu.Unlock() + return fmt.Errorf("change prompt profile: %w", err) + } + // Fresh slices leave snapshots held by an earlier request untouched. + a.tools = append([]bare.Tool(nil), tools...) + a.definitions = definitions + a.armMu.Unlock() + if next.lean() { + a.memoryText = "" + } + a.rerenderSystemLocked(time.Now()) + // The last usage figure counted the old prefix. The new request is weighed + // again by the provider after the new prompt and schemas are encoded. + a.contextTokens = 0 + a.contextBeltTokens = 0 + return nil +} diff --git a/internal/session/promptprofile_live_test.go b/internal/session/promptprofile_live_test.go new file mode 100644 index 0000000000..78acc3933a --- /dev/null +++ b/internal/session/promptprofile_live_test.go @@ -0,0 +1,146 @@ +package session + +import ( + "context" + "encoding/json" + "strings" + "testing" + + "github.com/Agent-Field/agentfield/sdk/go/ai" + "github.com/Agent-Field/codeaf/internal/exec/bare" +) + +func TestAutomaticProfileFollowsModelSwitchOnNextRequest(t *testing.T) { + t.Setenv(promptProfileEnv, "") + var agent *Agent + wantLean := true + script := &scriptedCompleter{} + check := func(_ context.Context, messages []ai.Message) (*ai.Response, error) { + if got := agent.config.promptProfile().lean(); got != wantLean { + t.Errorf("lean=%v, want %v", got, wantLean) + } + if hasSection := strings.Contains(messageContentText(messages[0]), "# Interrupts and steering"); hasSection == wantLean { + t.Error("prompt did not change with the tool belt") + } + if hasTask := agent.hasTool("propose_task"); hasTask == wantLean { + t.Error("default task schema did not follow the profile") + } + return textResponse("done"), nil + } + script.steps = []step{check, check} + agent, _ = newTestAgent(t, script, func(c *Config) { + buildShippedConversation(t, c) + c.Memory = nil + c.System = "" + c.Model = "test/large" + c.ContextWindowFor = func(model string) int { + if model == "test/small" { + return 16385 + } + return 128000 + } + }) + if agent.config.promptProfile().lean() { + t.Fatal("fixture did not start full") + } + for _, model := range []string{"test/small", "test/large"} { + wantLean = model == "test/small" + agent.SetModel(model) + for _, event := range collect(t, mustSubmit(t, agent, "hello")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + } +} + +func TestAutomaticProfilePreservesLoadedCapabilities(t *testing.T) { + t.Setenv(promptProfileEnv, "") + agent := leanShapedAgent(t) + if _, failed := agent.loadCapability("tasks"); failed { + t.Fatal("load tasks") + } + agent.SetContextWindow(128000) + if err := agent.preparePromptProfile(agent.Model()); err != nil { + t.Fatal(err) + } + if agent.config.promptProfile().lean() || !agent.hasTool("propose_task") || !agent.remembers() { + t.Fatal("full profile did not restore capabilities and memory") + } + agent.SetContextWindow(16385) + if err := agent.preparePromptProfile(agent.Model()); err != nil { + t.Fatal(err) + } + if !agent.config.promptProfile().lean() || !agent.hasTool("propose_task") || agent.remembers() || agent.hasTool("remember") { + t.Fatal("lean profile lost explicitly loaded tasks or retained memory") + } +} + +func TestAutomaticProfileHonorsExplicitPins(t *testing.T) { + for _, pin := range []string{"lean", "full"} { + t.Run(pin, func(t *testing.T) { + t.Setenv(promptProfileEnv, pin) + agent := leanShapedAgent(t) + agent.SetContextWindow(128000) + if err := agent.preparePromptProfile(agent.Model()); err != nil { + t.Fatal(err) + } + if got := string(agent.config.promptProfile()); got != pin { + t.Fatalf("profile=%s", got) + } + agent.SetContextWindow(6000) + if err := agent.preparePromptProfile(agent.Model()); err != nil { + t.Fatal(err) + } + if got := string(agent.config.promptProfile()); got != pin { + t.Fatalf("profile=%s", got) + } + }) + } +} + +func TestAutomaticProfileSwitchDuringToolTurn(t *testing.T) { + t.Setenv(promptProfileEnv, "") + var agent *Agent + script := &scriptedCompleter{steps: []step{ + func(context.Context, []ai.Message) (*ai.Response, error) { + return toolResponse("choose-1", "choose", `{}`), nil + }, + func(_ context.Context, messages []ai.Message) (*ai.Response, error) { + if !agent.config.promptProfile().lean() || strings.Contains(messageContentText(messages[0]), "# Interrupts and steering") { + t.Error("second request kept the large model's profile") + } + assertPaired(t, messages) + if !agent.hasTool("choose") { + t.Error("model switch dropped a dynamically armed tool") + } + return textResponse("done"), nil + }, + }} + agent, _ = newTestAgent(t, script, func(c *Config) { + c.System = "" + c.ContextWindowFor = func(model string) int { + if model == "test/small" { + return 6144 + } + return 128000 + } + }) + changes := 0 + _, err := agent.armFamily([]bare.Tool{{Name: "choose", Description: "selects the smaller model", Schema: json.RawMessage(`{"type":"object"}`), Execute: func(context.Context, json.RawMessage) (string, bool, error) { + changes++ + agent.SetModel("test/small") + return "selected", false, nil + }}}) + if err != nil { + t.Fatal(err) + } + for _, event := range collect(t, mustSubmit(t, agent, "choose")) { + if event.Kind == EventError { + t.Fatal(event.Err) + } + } + if changes != 1 || script.requests() != 2 { + t.Fatalf("changes=%d requests=%d", changes, script.requests()) + } +} diff --git a/internal/session/prompts/chat.md b/internal/session/prompts/chat.md new file mode 100644 index 0000000000..7b42e623dc --- /dev/null +++ b/internal/session/prompts/chat.md @@ -0,0 +1,26 @@ +You are codeaf: a colleague in a conversation. Help the person think, decide and +get useful answers. + +# What you can do here +This model has no tools in codeaf: you cannot read, search, write or run +anything, and you cannot look anything up. Answer from what you know and from +what the person pastes in. When a request needs files, commands or current +information, say so in one sentence and do what you can without them; the +person can switch to a model that uses tools with /model. + +# The answer +- Line one answers the question or states the outcome. Then the thing asked + for, in the form asked. Then, if needed, a few lines of why. +- An answer is a few sentences; a deliverable is as long as the work needs. + "Explain", "why" or "walk me through" lift the limit. +- Structure only where the content has it: a table for comparisons, numbered + steps for a sequence, prose otherwise. No emoji, no decorative bold. +- State uncertainty at the claim it affects. Never invent sources, quotes, + figures or recent events you cannot know. +- No opener, no recap, no closing offer. + +# When corrected +- A correction or restated ask is the new ask: deliver it. No apology, no + defense of the earlier answer. +- If you believe the person is factually wrong, give the evidence in one or two + lines, then still deliver what they asked. diff --git a/internal/session/recovery_law_test.go b/internal/session/recovery_law_test.go index cecefdbd9a..9659ca34fd 100644 --- a/internal/session/recovery_law_test.go +++ b/internal/session/recovery_law_test.go @@ -59,6 +59,7 @@ var boundsACount = map[string]string{ "DegenerateCutAttempts": "internal/taxonomy's, as above", "BlindCutAttempts": "internal/taxonomy's, as above", "programAutoRetries": "hand-offs of one piece of work to a program that codeaf starts on its own, each a new billed run on a sharper brief — the owner's cap on spending without them, not a patience for one call", + "contextRecoveryAttempts": "compaction rounds after a context refusal, each sending a SMALLER conversation and only when something shrank or was learned — never the refused request again", } // TestNoAttemptCountingLoopInTheSession refuses a loop that counts its own diff --git a/internal/session/session.go b/internal/session/session.go index 542339016c..737da345d8 100644 --- a/internal/session/session.go +++ b/internal/session/session.go @@ -87,21 +87,21 @@ const ( EventTurnDone // EventError ends the turn abnormally; Err says why. EventError - // EventCompacting says a compaction pass has started, which is work a - // surface should show rather than silence. Hint sizes the pass + // EventCompacting says a compaction pass has edited the transcript, which is + // work a surface should show rather than silence. Hint sizes the pass // ("compacting ~84k tokens"). // - // EventCompacted always follows it, success or failure — a surface opens a - // row on this one and settles it on that one, and a pass that found nothing - // to do says so rather than leaving the row open (loop.go's [Agent.compact]). - // There is no summarizer behind it any more: the pass is two mechanical - // walks over messages this session already holds, so what it costs is a lock - // and not a model call. + // EventCompacted always follows it — a surface opens a row on this one and + // settles it on that one. A pass that found nothing to do sends neither, so + // it opens no row that would need settling (loop.go's [Agent.compactWithPolicy]); + // `/compact` hears that from [ErrNothingToCompact] instead. Most passes are + // two mechanical walks over messages this session already holds, so what + // they cost is a lock; only a pass those walks cannot finish writes a summary + // with the conversation's model (compact_summary.go). EventCompacting // EventCompacted marks a compaction pass; Hint summarizes - // ("compacted from ~84k tokens, kept last ~20k"), and [Event.Unchanged] - // separates the pass that edited the transcript from the one that found - // nothing to do. + // ("compacted from ~84k tokens, kept last ~20k"). [Event.Unchanged] is only + // ever set on it by a peer built before a no-op pass went silent. EventCompacted // EventReasoning carries one streamed chunk of the model's REASONING in // Text, for the models that put their working on the wire (OpenRouter's @@ -705,6 +705,16 @@ type Event struct { // cannot support. Skills []string `json:"Skills,omitempty"` + // Summarized is how many messages a compaction pass replaced with a + // summary, on [EventCompacted]; zero for a pass that only stubbed and + // folded. It is the field a surface decides by, never the hint's words: + // a pass that rewrote the person's own messages is the one whose line + // stays standing (internal/tui3's workfold.go), and a free pass folds + // with the rest of the turn's machinery. Omitted when zero, so an event + // with no summary serialises exactly as it did before this field, and a + // surface talking to an older engine folds every pass. + Summarized int `json:"Summarized,omitempty"` + // Category is the FAMILY OF WORK an EventCaption's sentence is about — one // word from the closed list in actioncategory.go — and it is zero on every // other kind. @@ -739,6 +749,12 @@ type Event struct { // It rides the wire behind a json tag of its own, so a peer built before it // existed does not send it, reads false, and behaves exactly as it always // did (internal/remote embeds this struct whole). + // + // NO PASS IN THIS BUILD SETS IT. A pass that finds nothing now sends neither + // EventCompacting nor EventCompacted, so there is no row to settle. The field + // stays, and a surface still honours it, because a remote engine built + // between the two changes announces every pass and settles a refused one + // with it. Unchanged bool `json:"Unchanged,omitempty"` // Args is the tool call's arguments rendered for display: the JSON the @@ -2239,16 +2255,12 @@ type Config struct { // (promptprofile.go's [resolvePromptProfile] is the whole ladder). PromptProfile string - // profile is which of the two fixed prefixes this session sends, SETTLED - // ONCE by newAgent before anything is built from it (promptprofile.go). - // - // It is a field on the config rather than on the agent because everything - // that reads it reads it before the agent exists — the page is rendered - // first and the belt is built from the same config a moment later — which is - // the law beltfacts.go's predicates are already written under. Empty means - // nobody has settled it, and [Config.promptProfile] then derives the answer - // live, which is what a test asking the question of a bare Config wants. + // profile is the initial shape, settled before the page and belt exist. + // liveProfile follows model changes for automatic conversation profiles; + // neither field requires mutation of this shared Config after construction. profile promptProfile + // liveProfile publishes automatic window changes without mutating Config. + liveProfile *livePromptProfile } // Agent is one conversation. It is safe for concurrent use, but Submit @@ -2416,6 +2428,9 @@ type Agent struct { // already on its way onto the belt (tools_capabilities.go). Nil on every // shape that pre-arms nothing, which is every full-profile belt. prearm []bare.Tool + // profileArmed retains explicitly loaded tools across profile changes. + // Like the belt itself, this ordered list is guarded by armMu. + profileArmed []bare.Tool // withdrawn is the record of a belt narrowed ON PURPOSE (withdrawn.go): the // hands the harness took, why, and what is left. Nil whenever the belt is // whole, which is nearly always. @@ -3139,26 +3154,25 @@ type Agent struct { // cutPointLocked, which already holds the lock: a second acquisition there // would deadlock the one call — Interrupt — that must always be answerable. contextWindow atomic.Int64 - // servedWindow is what this process has LEARNED about the window the model - // now in use really has, as opposed to the one its catalog row claims: the - // narrowest prompt that model has been refused for being too long - // (internal/provider's ServedWindow). Zero means nothing has been learned - // and the claim stands alone, which is the ordinary case. - // - // It is a field rather than a call because the memo is keyed by MODEL and - // the model is guarded by mu, while the threshold is read from - // cutPointLocked with mu already held — so it is atomic for - // [Agent.contextWindow]'s reason, word for word, and refreshed wherever the - // model or the window moves. + // servedWindow is the explicit total context limit most recently learned + // in this session. It is atomic because fold helpers read it under mu; + // endpoint-scoped persistence belongs to the provider, not the model id. servedWindow atomic.Int64 // compacting serializes compaction passes. One pass reads the transcript, // releases the lock to summarize, then rebuilds; a second pass entering // that window would summarize a prefix the first one is about to drop. compacting bool + // compactDone wakes a refused turn waiting for that pass to finish. It is + // closed under the same lock that clears compacting, so no wake is missed. + compactDone chan struct{} // contextTokens is the last provider-reported context size, the honest // figure when there is one. Zero means "estimate from content". contextTokens int + // contextBeltTokens belongs only to a rebuilt window after compaction. It + // keeps later transcript growth on the same footing until a provider count + // replaces the estimate. + contextBeltTokens int // followups is the second injection queue (agent.go). Steering drains at a // step boundary INTO the running turn; a follow-up waits for the turn to diff --git a/internal/session/sessionfile.go b/internal/session/sessionfile.go index 410283c935..84cdb84f31 100644 --- a/internal/session/sessionfile.go +++ b/internal/session/sessionfile.go @@ -229,6 +229,9 @@ type sessionEntry struct { // lines and reconstructs the transcript verbatim rather than from counts. Stubbed int `json:"stubbed,omitempty"` Folded int `json:"folded,omitempty"` + // Summarized is how many messages a summary note replaced + // (compact_summary.go). The note itself is in the window like any line. + Summarized int `json:"summarized,omitempty"` // Window is HOW MANY MESSAGE LINES THE PASS RE-JOURNALED BEHIND THIS MARKER // — the length of the rebuilt window [sessionFile.appendCompaction] writes @@ -2833,6 +2836,7 @@ func (s *sessionFile) appendCompaction(pass compactionPass, tokensBefore int, wi TokensBefore: tokensBefore, Stubbed: pass.stubbed, Folded: pass.folded, + Summarized: pass.summarized, // The length is written BEFORE the window it describes, which is the only // order that survives a crash halfway through: a reader that finds fewer // lines than the number promised has a truncated file and can say so, diff --git a/internal/session/taxonomy_boundary.go b/internal/session/taxonomy_boundary.go index e0a5b3ee54..f15fcb4fbb 100644 --- a/internal/session/taxonomy_boundary.go +++ b/internal/session/taxonomy_boundary.go @@ -429,11 +429,8 @@ func (a *Agent) readEmptyReply(model string, attempt int, outOfTime bool) taxono // wire, the model or the job — it is the REQUEST, and the answer is the same // question in fewer words ([taxonomy.Shape], [taxonomy.ActionCompact]). // -// `compacted` is whether this turn has already paid for that once, and it is the -// caller's fact because only the turn knows. The policy is what decides what to -// do with it: a second overflow after a compaction is a request that is not -// going to fit, and it comes back as work rather than as another round of the -// same move. +// `compacted` says the pending generation exhausted its recovery allowance. +// Only the caller can know that; successful work resets the allowance. func (a *Agent) readOverflow(err error, model string, compacted bool) taxonomy.Verdict { evidence := wireEvidence(err, 1) if !evidence.Overflow { diff --git a/internal/session/tools.go b/internal/session/tools.go index 5115052a8c..3edaa13774 100644 --- a/internal/session/tools.go +++ b/internal/session/tools.go @@ -152,6 +152,12 @@ func (a *Agent) belt() []bare.Tool { if a.config.mayBashBelt() { return a.bashBelt() } + // A MODEL WITH NO TOOLS CARRIES NO BELT, and no shelf a loader could reach + // either: the catalog says it cannot call one (chatpage.go). + if a.config.promptProfile().chat() { + a.clearShelf() + return nil + } tools := bare.AllToolsCapped(a.config.Workspace, a.resultCaps()) for index, tool := range tools { switch tool.Name { diff --git a/internal/session/turnfold.go b/internal/session/turnfold.go index a07532a4ef..febd9b7d15 100644 --- a/internal/session/turnfold.go +++ b/internal/session/turnfold.go @@ -80,7 +80,7 @@ type turnFoldReplacement struct { // range it held (turnFoldReadRepeats); every other result keeps the filed // pointer it always got. func (a *Agent) foldTurnOutputs(seenThrough int, consumedReads map[*ai.ToolCall]bool, hub *eventHub) { - line := turnWorkingSet(a.window()) + line := turnWorkingSet(a.trustedWindow()) if line <= 0 { return } @@ -98,7 +98,7 @@ func (a *Agent) foldTurnOutputs(seenThrough int, consumedReads map[*ai.ToolCall] limit := a.turnFoldLimitLocked(seenThrough) batches := turnFoldBatches(a.messages, a.turnFloor, limit, consumedReads) keepNewest, _ := turnFoldReadRepeats(a.messages, batches) - selected = turnFoldSelection(a.messages, batches, keepNewest, total, turnWorkingTarget(a.window())*bytesPerToken) + selected = turnFoldSelection(a.messages, batches, keepNewest, total, turnWorkingTarget(a.trustedWindow())*bytesPerToken) } a.mu.Unlock() if total <= line*bytesPerToken || selected == 0 { @@ -127,7 +127,7 @@ func (a *Agent) foldTurnOutputs(seenThrough int, consumedReads map[*ai.ToolCall] earlier := shapeEntries(a.messages, a.file, a.presentation) place := a.resultPlaceLocked() - target := turnWorkingTarget(a.window()) * bytesPerToken + target := turnWorkingTarget(a.trustedWindow()) * bytesPerToken batches := turnFoldBatches(a.messages, a.turnFloor, limit, consumedReads) keepNewest, superseded := turnFoldReadRepeats(a.messages, batches) // A pass that cannot buy the whole headroom does not run. Every rewrite @@ -214,6 +214,7 @@ func (a *Agent) foldTurnOutputs(seenThrough int, consumedReads map[*ai.ToolCall] note := textMessage("user", marker) a.messages = append(a.messages, note) a.contextTokens = 0 + a.contextBeltTokens = 0 // The original transcript becomes the scroll-back region, and the rebuilt // window is journaled behind a compaction marker exactly as the cross-turn diff --git a/internal/taxonomy/policy.go b/internal/taxonomy/policy.go index ada692952c..52c44caf8c 100644 --- a/internal/taxonomy/policy.go +++ b/internal/taxonomy/policy.go @@ -534,11 +534,9 @@ func (shapePolicy) Class() Class { return Shape } func (shapePolicy) Decide(e Evidence, _ Limits) Verdict { if e.Overflow { - // COMPACTION IS OFFERED ONCE PER TURN AND THEN NEVER AGAIN. A request - // that still does not fit after it was made smaller is not going to fit, - // and offering the same move a second time is a loop with a person - // watching it. The second one comes back as WORK — the honest verdict, - // because there is nothing left for this package to sell. + // The caller reports an exhausted recovery episode, not whether + // anything earlier in a long turn was ever compacted. Successful + // generations reset the episode; identical failed requests do not. if e.Compacted { return Verdict{Class: Work, Action: ActionReport, Reason: ReasonTooBig} } diff --git a/internal/taxonomy/taxonomy.go b/internal/taxonomy/taxonomy.go index 2872cf0a07..eacd9a7ab0 100644 --- a/internal/taxonomy/taxonomy.go +++ b/internal/taxonomy/taxonomy.go @@ -263,10 +263,8 @@ type Evidence struct { // envelope code ([provider.APIError.Overflow]). Overflow bool - // Compacted says this turn has ALREADY made the request smaller once. It is - // the difference between [ActionCompact] and giving the overflow back as - // [Work]: compaction that did not fit is a request that is not going to fit, - // and asking for it twice is a loop. + // Compacted says this failed generation exhausted its bounded context + // recovery allowance. Successful generations begin a new episode. Compacted bool // Spent says the TRANSPORT'S OWN SHAPE LADDER has been climbed and has run diff --git a/internal/tui2/tokens/glyph.go b/internal/tui2/tokens/glyph.go index 58233d8855..8d511d1098 100644 --- a/internal/tui2/tokens/glyph.go +++ b/internal/tui2/tokens/glyph.go @@ -200,6 +200,9 @@ const ( GlyphFilter = "⌕" // narrowing what is already on the page GlyphWrite = "✎" // a call that wrote something down + // GlyphCompacted marks a pass that shortened the model's working context. + GlyphCompacted = "⚭" + // The action families (internal/tui3's step gutter). One still, monochrome // mark per FAMILY of work — searching, editing, running a command — keyed // off the closed vocabulary the engine carries in session.ActionCategory. @@ -480,6 +483,7 @@ func Glyphs() []GlyphInfo { {"Search", GlyphSearch, '⌕', false}, {"Filter", GlyphFilter, '⌕', false}, {"Write", GlyphWrite, '✎', false}, + {"Compacted", GlyphCompacted, '⚭', false}, {"ActionRead", GlyphActionRead, '▤', true}, {"ActionCreate", GlyphActionCreate, '+', false}, {"ActionTest", GlyphActionTest, '◎', true}, diff --git a/internal/tui2/tokens/glyph_test.go b/internal/tui2/tokens/glyph_test.go index 3384d8e04a..25ec902653 100644 --- a/internal/tui2/tokens/glyph_test.go +++ b/internal/tui2/tokens/glyph_test.go @@ -247,7 +247,7 @@ func TestGlyphInventoryIsComplete(t *testing.T) { GlyphScopeUp, GlyphPointer, GlyphRecommended, GlyphFrameTopLeft, GlyphFrameTopRight, GlyphFrameBottomLeft, GlyphFrameBottomRight, GlyphFrameEdge, GlyphFrameSide, GlyphFrameTeeDown, GlyphFrameTeeUp, GlyphTarget, GlyphTruncated, GlyphCut, GlyphPromptChat, GlyphPromptSteer, GlyphReplyIn, - GlyphThought, GlyphShell, GlyphSearch, GlyphWrite, + GlyphThought, GlyphShell, GlyphSearch, GlyphWrite, GlyphCompacted, GlyphBoosted, GlyphSeparator, GlyphMissing, GlyphEstimate, GlyphAccentRail, GlyphHugEdge, GlyphChipCapLeft, GlyphChipCapRight, GlyphDragHandle, GlyphStepDone, GlyphStepRunning, diff --git a/internal/tui2/tokens/glyphset.go b/internal/tui2/tokens/glyphset.go index dec2ef3f3d..0c161febd2 100644 --- a/internal/tui2/tokens/glyphset.go +++ b/internal/tui2/tokens/glyphset.go @@ -135,6 +135,7 @@ const ( GShell GSearch GWrite + GCompacted GActionRead GActionCreate GActionTest diff --git a/internal/tui2/tokens/nerdfont.go b/internal/tui2/tokens/nerdfont.go index 4b3ccfc63f..f2ff272e4f 100644 --- a/internal/tui2/tokens/nerdfont.go +++ b/internal/tui2/tokens/nerdfont.go @@ -367,6 +367,12 @@ var vocabulary = []GlyphBinding{ ASCII: "*", UsualTint: Cyan, NFAmbiguous: true, AutoUpgrade: true, }, + { + ID: GCompacted, Name: "Compacted", Meaning: "a pass shortened the model's working context", + Plain: GlyphCompacted, NerdFont: "\uF066", NFName: "nf-fa-compress", + ASCII: "#", + UsualTint: TextTertiary, NFAmbiguous: true, AutoUpgrade: true, + }, // -- the action families (internal/tui3's step gutter) ------------------- // diff --git a/internal/tui2/tokens/testdata/nerdfont_glyphnames.json b/internal/tui2/tokens/testdata/nerdfont_glyphnames.json index b82a3dd6dc..1eee70a012 100644 --- a/internal/tui2/tokens/testdata/nerdfont_glyphnames.json +++ b/internal/tui2/tokens/testdata/nerdfont_glyphnames.json @@ -47,6 +47,9 @@ "nf-fa-code_fork": { "code": "f126" }, + "nf-fa-compress": { + "code": "f066" + }, "nf-fa-cog": { "code": "f013" }, diff --git a/internal/tui3/app.go b/internal/tui3/app.go index 5294aaaee7..a8f6563a5a 100644 --- a/internal/tui3/app.go +++ b/internal/tui3/app.go @@ -2,6 +2,7 @@ package tui3 import ( "context" + "errors" "fmt" "os" "os/exec" @@ -20,6 +21,7 @@ import ( "github.com/Agent-Field/codeaf/internal/credits" internalenv "github.com/Agent-Field/codeaf/internal/env" "github.com/Agent-Field/codeaf/internal/modelsource" + "github.com/Agent-Field/codeaf/internal/remote" "github.com/Agent-Field/codeaf/internal/session" "github.com/Agent-Field/codeaf/internal/skills" "github.com/Agent-Field/codeaf/internal/subharness" @@ -57,6 +59,10 @@ const markdownThrottle = 1500 * time.Millisecond // change because the frames arrived over a wire (link.go's [app.dueEvery]). const usageEvery = 10 +// compactStillRunning is what /compact says when the engine is still working +// after the surface's wait ran out: the pass lands on its own. +const compactStillRunning = "still compacting — it is taking longer than usual and finishes on its own; the token count in the status line drops when it lands" + // quietBeforeEllipsis is how long the stream has to be silent before the // ellipsis appears under a reply that is already streaming. Text arriving in // chunks a few hundred milliseconds apart is a working model, not a stalled @@ -301,6 +307,10 @@ type entry struct { // `@deepseek` gone from the model word and `▸ worked 1.6s · ctrl+e` where // the explanation should have been (session's EventRowNews). told bool + // summarized says an [entryCompact] pass replaced some of the conversation + // with a summary ([session.Event.Summarized]). Only such a pass stands + // outside the turn's fold (workfold.go); a free one folds with the work. + summarized bool // carried marks the note naming the skills a turn carried (session's // turnSkillsNotice). It is not addressed to the person, so it does not hold @@ -680,7 +690,11 @@ type ( lump bool } streamClosedMsg struct{ gen int } - compactedMsg struct{ err error } + compactedMsg struct { + err error + before, after int + agent Agent + } // frameMsg is the paint clock: it promotes whatever streamed since the // last one into a frame, and steps the animations. frameMsg struct{} @@ -5154,9 +5168,61 @@ func (a *app) route(msg tea.Msg) (tea.Model, tea.Cmd) { return a, a.steerFell(msg) case compactedMsg: - if msg.err != nil { - a.toldNote("compact failed: " + msg.err.Error()) + // A LATE PASS BELONGS TO THE CONVERSATION THAT ASKED FOR IT. A + // page over that conversation still leaves its reply behind the page. + if msg.agent != nil && msg.agent != a.agent { + return a, nil + } + if why, skipped := session.SummarySkippedWhy(msg.err); skipped { + lead, separator := a.icon(tokens.GCompacted), " · " + if a.linear || a.pal.ascii { + separator = " - " + why = compactASCII(why) + } + line := lead + " compacted" + if msg.before > msg.after && msg.after > 0 { + line += fmt.Sprintf("%sabout %d to %d tokens", separator, msg.before, msg.after) + } + a.toldNote(line + separator + "summary skipped: " + why) + a.measureContext() + a.noticeEvent(eventCompacted) + } else if msg.err != nil { + if why, nothing := session.NothingToCompactWhy(msg.err); nothing { + // THE NO-OP SAYS WHY when the engine knows: too little older + // conversation to summarize, or the summary that would have + // shortened it did not land. An engine that says nothing more + // (a peer built before the reason existed) gets the old line. + if why == "" { + why = "your messages and recent work are kept" + } + a.toldNote("nothing to compact — " + why) + } else if errors.Is(msg.err, remote.ErrLate) { + // A PASS THAT OUTLIVED THE WAIT IS STILL RUNNING. The engine + // holds it, not this window, and it lands with the status + // line's count dropping; "failed" was the sentence for a pass + // that then succeeded (internal/remote's [Agent.Compact]). + a.toldNote(compactStillRunning) + } else { + a.toldNote("compact failed: " + msg.err.Error()) + } } else { + // THE SAME MARK AS A PASS THE ENGINE RAN ON ITS OWN ([app.divider]), + // so a person reading back can tell a compaction from any other note + // whichever door started it. + lead, separator := a.icon(tokens.GCompacted), " · " + if a.linear || a.pal.ascii { + separator = " - " + } + if msg.before > msg.after && msg.after > 0 { + a.toldNote(fmt.Sprintf("%s compacted%sabout %d to %d tokens", lead, separator, msg.before, msg.after)) + } else { + a.toldNote(lead + " compacted") + } + // THE METER FOLLOWS THE PASS, as it does for one inside a turn + // ([app.applyEvent]'s EventCompacted). Without this the status line + // kept the last request's weight — 585.1k over a conversation /compact + // had just taken to 15k — until the next message was sent. + a.measureContext() a.noticeEvent(eventCompacted) } return a, nil @@ -6057,13 +6123,15 @@ func (a *app) applyEvent(ev session.Event, lump bool) tea.Cmd { // the place over into the region the pass just created, so the history // stays reachable and stays in order. // - // AND ONLY FOR A PASS THAT ACTUALLY HAPPENED. The event is sent on both - // paths, so this used to hand the bookkeeping over on a pass that found - // nothing to stub and nothing to fold: replayFrom was dropped to a floor - // the reader was nowhere near, the seam was marked drawn without being - // drawn, and the conversation between the two went quiet. The surface - // then said there was nothing above it. Nothing had moved, so there is - // nothing to carry over ([session.Event.Unchanged]). + // AND ONLY FOR A PASS THAT ACTUALLY HAPPENED. A local engine no longer + // sends this event for a pass that found nothing, but a remote engine + // built before that change sends it on both paths. This used to hand the + // bookkeeping over on a pass that found nothing to stub and nothing to + // fold: replayFrom was dropped to a floor the reader was nowhere near, + // the seam was marked drawn without being drawn, and the conversation + // between the two went quiet. The surface then said there was nothing + // above it. Nothing had moved, so there is nothing to carry over + // ([session.Event.Unchanged]). if !ev.Unchanged { a.rebase() } @@ -7884,7 +7952,11 @@ func (a *app) slash(line string) tea.Cmd { case "compact": agent, ctx := a.agent, a.ctx a.note("compacting…") - return func() tea.Msg { return compactedMsg{err: agent.Compact(ctx)} } + return func() tea.Msg { + before := agent.ContextTokens() + err := agent.Compact(ctx) + return compactedMsg{err: err, before: before, after: agent.ContextTokens(), agent: agent} + } case "rewind": // THE COMMAND IS THE DELIBERATE DOOR AND IT OPENS THE TIMELINE diff --git a/internal/tui3/bundle_test.go b/internal/tui3/bundle_test.go index 44117584cd..ef9cb49387 100644 --- a/internal/tui3/bundle_test.go +++ b/internal/tui3/bundle_test.go @@ -1511,8 +1511,8 @@ func TestTheCompactionRowRunsAndThenSettlesInPlace(t *testing.T) { t.Fatalf("the settle added a row: %d compaction rows, want one", n) } settled := findRow(t, a, "compacted from ~168k tokens") - if !strings.Contains(settled, "⚭") || !strings.Contains(settled, "──") { - t.Fatalf("a finished pass is not the divider: %q", settled) + if !strings.Contains(settled, "⚭") || strings.Contains(settled, "──") { + t.Fatalf("a finished pass is not the one quiet compaction line: %q", settled) } if !strings.Contains(settled, "· took 6s") { t.Fatalf("the finished pass does not say what it took: %q", settled) @@ -1541,8 +1541,8 @@ func TestACompactedEventWithNothingRunningIsBornSettled(t *testing.T) { a.touch() row := findRow(t, a, "compacted from ~84k tokens") - if !strings.Contains(row, "⚭") || !strings.Contains(row, "──") { - t.Fatalf("the replayed pass is not the divider: %q", row) + if !strings.Contains(row, "⚭") || strings.Contains(row, "──") { + t.Fatalf("the replayed pass is not the one quiet compaction line: %q", row) } if strings.Contains(row, "took") { t.Fatalf("a pass nobody watched claimed a duration: %q", row) diff --git a/internal/tui3/commands.go b/internal/tui3/commands.go index 08f077020a..bdf3e7b68f 100644 --- a/internal/tui3/commands.go +++ b/internal/tui3/commands.go @@ -82,7 +82,7 @@ var commands = []command{ // on it rather than on "unknown command: /clear". {name: "new", desc: "start another conversation in this project", alias: []string{"clear", "clean", "reset"}}, {name: "resume", desc: "open an earlier conversation", alias: []string{"sessions"}}, - {name: "compact", desc: "summarize the conversation now"}, + {name: "compact", desc: "shorten the conversation now"}, {name: "drafts", desc: "cleared-but-kept drafts · enter restores one, d lets one go"}, {name: "stop", desc: "stop the open task or selected work · asks first"}, {name: "autonomy", desc: "how questions are handled while you are away"}, diff --git a/internal/tui3/compact_ascii_words_test.go b/internal/tui3/compact_ascii_words_test.go new file mode 100644 index 0000000000..ef7969d3e1 --- /dev/null +++ b/internal/tui3/compact_ascii_words_test.go @@ -0,0 +1,23 @@ +package tui3 + +import ( + "strings" + "testing" +) + +// The linear tier is a screen reader's: a compaction line's marks become ASCII, +// but a reason or a summary's first line in the person's own language is read +// out as written, never as a row of question marks. +func TestCompactLineOnTheLinearTierKeepsWordsInAnyScript(t *testing.T) { + got := compactASCII("summary skipped: résumé terminé — 要約 · ~7k → ~3k tokens…") + for _, word := range []string{"résumé", "terminé", "要約"} { + if !strings.Contains(got, word) { + t.Fatalf("linear hint lost %q: %q", word, got) + } + } + for _, mark := range []string{"—", "·", "→", "…"} { + if strings.Contains(got, mark) { + t.Fatalf("linear hint still carries the mark %q: %q", mark, got) + } + } +} diff --git a/internal/tui3/compact_command_test.go b/internal/tui3/compact_command_test.go new file mode 100644 index 0000000000..1083cbb167 --- /dev/null +++ b/internal/tui3/compact_command_test.go @@ -0,0 +1,246 @@ +package tui3 + +import ( + "errors" + "strings" + "testing" + "time" + + "github.com/Agent-Field/codeaf/internal/config" + "github.com/Agent-Field/codeaf/internal/remote" + "github.com/Agent-Field/codeaf/internal/session" + "github.com/Agent-Field/codeaf/internal/tui2/tokens" +) + +func TestCompactCommandReportsReductionAndExplainsProtectedHistory(t *testing.T) { + for _, test := range []struct { + message compactedMsg + want string + }{ + {compactedMsg{before: 60000, after: 5000}, "compacted · about 60000 to 5000 tokens"}, + {compactedMsg{err: session.ErrNothingToCompact}, "nothing to compact — your messages and recent work are kept"}, + {compactedMsg{err: errors.New(session.ErrNothingToCompact.Error())}, "nothing to compact — your messages and recent work are kept"}, + // THE NO-OP SAYS WHY, from this engine and from a remote one whose + // error arrives as its words alone. + {compactedMsg{err: &session.NothingToCompact{Why: "only ~400 tokens since the last summary — too little to summarize"}}, + "nothing to compact — only ~400 tokens since the last summary — too little to summarize"}, + {compactedMsg{err: errors.New("session: nothing to compact: the summary was interrupted")}, + "nothing to compact — the summary was interrupted"}, + {compactedMsg{err: &session.SummarySkipped{Why: "provider unavailable"}, before: 60000, after: 5000}, + "compacted · about 60000 to 5000 tokens · summary skipped: provider unavailable"}, + {compactedMsg{err: errors.New("session: compacted: summary skipped: provider unavailable"), before: 60000, after: 60000}, + "compacted · summary skipped: provider unavailable"}, + // A PASS THAT OUTLIVED THE WAIT IS STILL RUNNING, never "failed". + {compactedMsg{err: remote.ErrLate}, "still compacting — it is taking longer than usual and finishes on its own"}, + {compactedMsg{err: errors.New("the model refused")}, "compact failed: the model refused"}, + } { + a := newTestApp(&fakeAgent{model: "m"}) + drive(t, a, test.message) + if got := lastNote(t, a); !strings.Contains(got, test.want) { + t.Fatalf("got %q, want %q", got, test.want) + } + } +} + +func TestCompactReplyFromAnotherConversationChangesNeitherPageNorMeter(t *testing.T) { + first := &fakeAgent{model: "first", weight: 8615} + a := newTestApp(first) + cmd := a.slash("/compact") + if cmd == nil { + t.Fatal("/compact did not start") + } + second := &fakeAgent{model: "second", weight: 9000} + a.takeUp(Conversation{Agent: second}, false) + a.entries = nil + a.ctxTokens = 1234 + delete(a.notices.seen, eventCompacted) + drive(t, a, cmd()) + if len(a.entries) != 0 || a.ctxTokens != 1234 || a.notices.seen[eventCompacted] { + t.Fatalf("late reply changed the new conversation: entries=%v, meter=%d, compact notice=%v", a.entries, a.ctxTokens, a.notices.seen[eventCompacted]) + } +} + +func TestCompactReplyOnItsOriginalConversationStillReportsSuccess(t *testing.T) { + a := newTestApp(&fakeAgent{model: "first", weight: 8615}) + cmd := a.slash("/compact") + if cmd == nil { + t.Fatal("/compact did not start") + } + drive(t, a, cmd()) + if got := lastNote(t, a); !strings.Contains(got, "compacted") { + t.Fatalf("the pass was not reported in its original conversation: %q", got) + } + if a.ctxTokens != 8615 || !a.notices.seen[eventCompacted] { + t.Fatalf("the pass did not update its original conversation: meter=%d, notice=%v", a.ctxTokens, a.notices.seen[eventCompacted]) + } +} + +func TestCompactReplyBehindHomeStaysWithOriginalConversation(t *testing.T) { + a := newTestApp(&fakeAgent{model: "first", weight: 8615}) + cmd := a.slash("/compact") + if cmd == nil { + t.Fatal("/compact did not start") + } + a.openHome() + if !a.at(pageHome) { + t.Fatal("Home did not open") + } + a.entries = nil + a.ctxTokens = 1234 + delete(a.notices.seen, eventCompacted) + drive(t, a, cmd()) + if len(a.entries) != 1 || a.ctxTokens != 8615 || !a.notices.seen[eventCompacted] { + t.Fatalf("original conversation lost reply behind Home: entries=%v, meter=%d, compact notice=%v", a.entries, a.ctxTokens, a.notices.seen[eventCompacted]) + } + a.closeHome() + if got := lastNote(t, a); !strings.Contains(got, "compacted") { + t.Fatalf("reply missing on return: %q", got) + } +} + +func TestCompactReplyReportsSizeOnlyWhenTheContextShrinks(t *testing.T) { + for _, size := range []int{8615, 9000} { + a := newTestApp(&fakeAgent{model: "m"}) + drive(t, a, compactedMsg{before: 8615, after: size}) + if got := lastNote(t, a); strings.Contains(got, "about") { + t.Fatalf("size %d was described as a reduction: %q", size, got) + } + } +} + +func TestCompactLinesUseOnlyASCIIOnTheLinearTier(t *testing.T) { + a := newTestApp(&fakeAgent{model: "m"}) + a.linear = true + line := a.divider("compacted · summarized 4 messages · ~31k → ~13k tokens · full record in the session journal", 120) + drive(t, a, compactedMsg{before: 60000, after: 5000}) + line += lastNote(t, a) + line += a.divider("compacted · summary skipped: the model refused … "+strings.Repeat("a", 150), 120) + for _, r := range line { + if r > 127 { + t.Fatalf("linear compaction line contains non-ASCII %q: %q", r, line) + } + } + b := newTestApp(&fakeAgent{model: "m"}) + b.pal.ascii = true + drive(t, b, compactedMsg{err: &session.SummarySkipped{Why: "the model refused …"}, before: 60000, after: 5000}) + line = b.divider("compacted · summary skipped: the model refused …", 120) + lastNote(t, b) + for _, r := range line { + if r > 127 { + t.Fatalf("ASCII compaction line contains non-ASCII %q: %q", r, line) + } + } +} + +func TestCompactLinesUseTheSelectedGlyphTier(t *testing.T) { + for _, set := range []tokens.GlyphSet{tokens.Plain, tokens.NerdFont} { + a := newTestApp(&fakeAgent{model: "m"}) + a.pal.icons = set + mark := set.Glyph(tokens.GCompacted) + if got := a.divider("compacted", 40); !strings.Contains(got, mark+" compacted") { + t.Errorf("%s divider = %q, missing selected mark %q", set, got, mark) + } + drive(t, a, compactedMsg{before: 60000, after: 5000}) + if got := lastNote(t, a); !strings.HasPrefix(got, mark+" compacted") { + t.Errorf("%s command note = %q, missing selected mark %q", set, got, mark) + } + } +} + +// A SUMMARY IN THE MIDDLE OF A TURN OUTLIVES THE FOLD. The work on either +// side of it goes behind the chip as it always did; the one quiet line saying +// the person's words were summarized stays, above the answer. +func TestACompactionMidTurnStaysVisibleWhenTheWorkFolds(t *testing.T) { + agent := &fakeAgent{model: "m", turns: [][]session.Event{{ + toolBegin("bash", "go build ./..."), + {Kind: session.EventToolEnd, Tool: "bash"}, + {Kind: session.EventCompacting, Hint: "compacting ~31k tokens"}, + {Kind: session.EventCompacted, Hint: "compacted · summarized 4 messages · ~31k → ~13k tokens", Summarized: 4}, + toolBegin("bash", "go test ./..."), + {Kind: session.EventToolEnd, Tool: "bash"}, + text(session.EventTextDelta, "all green"), + {Kind: session.EventTurnDone}, + }}} + a := newTestApp(agent) + runTurn(t, a, agent, "build and test it") + + page := plain(frame(a)) + if !strings.Contains(page, "· ⚭ compacted · summarized 4 messages") { + t.Fatalf("the compaction left no trace once the turn folded:\n%s", page) + } + if !strings.Contains(page, "all green") { + t.Fatalf("the answer is not standing:\n%s", page) + } + if strings.Contains(page, "go test ./...") || strings.Contains(page, "go build ./...") { + t.Fatalf("the work around the compaction did not fold:\n%s", page) + } +} + +// AN END-OF-TURN SUMMARY STAYS TOO. The check after the answer is the +// ordinary place a pass runs, and the fold that takes a turn's trailing +// bookkeeping into its disclosure must leave a summary's line standing under +// the answer, with the answer itself still in view. +func TestAnEndOfTurnCompactionStaysVisibleUnderTheAnswer(t *testing.T) { + a := newTestApp(&fakeAgent{model: "m"}) + a.entries = []entry{ + {kind: entryUser, text: "q", turn: 1}, + {kind: entryTool, tool: "read", status: toolOK, turn: 1, settled: true}, + {kind: entryAssistant, text: "The answer.", turn: 1, settled: true}, + {kind: entryCompact, text: "compacted · summarized 3 messages", summarized: true, turn: 1, began: time.Unix(90, 0), ended: time.Unix(100, 0)}, + } + a.workMode = config.WorkFold + a.touch() + text := strings.Join(plainRows(a), "\n") + answer, mark := strings.Index(text, "The answer."), strings.Index(text, "⚭ compacted · summarized 3 messages") + if answer < 0 || mark < 0 || mark < answer { + t.Fatalf("want the answer and then the summary line under it:\n%s", text) + } +} + +// A FREE PASS FOLDS WITH THE WORK, wherever it ran. On a small window a pass +// that only stubs and folds runs almost every step, and a standing line for +// each drew five marks between five chips on one turn (review of #1658). Only +// a pass that summarized stands; this one is behind the chip. +func TestAFreePassFoldsWithTheWorkAroundIt(t *testing.T) { + agent := &fakeAgent{model: "m", turns: [][]session.Event{{ + toolBegin("bash", "go build ./..."), + {Kind: session.EventToolEnd, Tool: "bash"}, + {Kind: session.EventCompacting, Hint: "compacting ~31k tokens"}, + {Kind: session.EventCompacted, Hint: "compacted · folded 6 messages · ~31k → ~24k tokens"}, + toolBegin("bash", "go test ./..."), + {Kind: session.EventToolEnd, Tool: "bash"}, + text(session.EventTextDelta, "all green"), + {Kind: session.EventCompacted, Hint: "compacted · folded 2 messages · ~25k → ~23k tokens"}, + {Kind: session.EventTurnDone}, + }}} + a := newTestApp(agent) + runTurn(t, a, agent, "build and test it") + + page := plain(frame(a)) + if strings.Contains(page, "folded 6 messages") || strings.Contains(page, "folded 2 messages") { + t.Fatalf("a free pass is standing outside the fold:\n%s", page) + } + if !strings.Contains(page, "all green") || strings.Count(page, "▸ worked") != 1 { + t.Fatalf("want one chip and the answer:\n%s", page) + } + drive(t, a, key("ctrl+e")) + if opened := plain(frame(a)); !strings.Contains(opened, "folded 6 messages") || !strings.Contains(opened, "folded 2 messages") { + t.Fatalf("the opened work does not carry the free passes:\n%s", opened) + } +} + +// THE METER FOLLOWS A /compact. The status line read the last request's weight +// until the next message, so a conversation just taken from 568k to 15k still +// showed 585.1k (sandbox #3, 2026-09-28). +func TestTheMeterDropsAsSoonAsACompactLands(t *testing.T) { + fake := &fakeAgent{model: "m", weight: 585_100} + a := newTestApp(fake) + a.measureContext() + if a.ctxTokens != 585_100 { + t.Fatalf("fixture meter = %d", a.ctxTokens) + } + fake.weight = 15_110 + drive(t, a, compactedMsg{before: 567_975, after: 15_110}) + if a.ctxTokens != 15_110 { + t.Fatalf("meter after /compact = %d, want the new weight 15110", a.ctxTokens) + } +} diff --git a/internal/tui3/feed.go b/internal/tui3/feed.go index 63cc0bf348..2be21f50f9 100644 --- a/internal/tui3/feed.go +++ b/internal/tui3/feed.go @@ -381,7 +381,7 @@ func (f *feed) ingestStream(ev session.Event, lump bool) { // spinning over a turn that moved on is the defect the pair exists to // close. f.closeLive() - f.settleCompaction(firstNonEmpty(ev.Hint, "compacted")) + f.settleCompaction(firstNonEmpty(ev.Hint, "compacted"), ev.Summarized > 0) } } @@ -930,13 +930,13 @@ func (f *feed) finishTool(ev session.Event) { // an older session predates the start event entirely. Those get a row born // finished — the divider they always drew, with no duration claimed, because a // pass this surface did not see the start of has no honest elapsed time. -func (f *feed) settleCompaction(text string) { +func (f *feed) settleCompaction(text string, summarized bool) { for i := len(f.entries) - 1; i >= 0; i-- { e := &f.entries[i] if e.kind != entryCompact || !e.ended.IsZero() { continue } - e.text, e.ended = text, f.now() + e.text, e.ended, e.summarized = text, f.now(), summarized e.stale = true // AND THE PAGE IS TOLD, HERE. Ingest is the whole of what an event does // (see [feed.ingest]), so a settle that left the repaint to its caller was @@ -947,7 +947,7 @@ func (f *feed) settleCompaction(text string) { } now := f.now() f.entries = append(f.entries, entry{ - kind: entryCompact, text: text, turn: f.turn, began: now, ended: now, + kind: entryCompact, text: text, turn: f.turn, began: now, ended: now, summarized: summarized, }) f.follow() f.touch() diff --git a/internal/tui3/notice.go b/internal/tui3/notice.go index 55c7e6a615..f2eadbb73e 100644 --- a/internal/tui3/notice.go +++ b/internal/tui3/notice.go @@ -340,7 +340,7 @@ var notices = []notice{ pct, ok := a.ctxPercent() return ok && pct >= contextHintPct }, - text: "/compact summarizes the conversation now", + text: "/compact shortens the conversation now", retire: eventCompacted, }, { diff --git a/internal/tui3/render.go b/internal/tui3/render.go index 55d000e63e..25e3f96071 100644 --- a/internal/tui3/render.go +++ b/internal/tui3/render.go @@ -4,6 +4,7 @@ import ( "path/filepath" "strings" "time" + "unicode" "github.com/charmbracelet/x/ansi" @@ -1861,17 +1862,39 @@ func (a *app) silentFor() time.Duration { return time.Since(a.lastDelta) } -// divider is the compaction mark: a rule with the fact in it, because a +var compactASCIIPunctuation = strings.NewReplacer(" · ", " - ", " → ", " to ", " — ", " - ", "…", "...") + +// compactASCII spells a compaction line's MARKS in ASCII and leaves its WORDS +// alone. The linear tier is a screen reader's, and a reason or a summary's first +// line in the person's own language is read out as written; only punctuation +// and symbols a plain terminal cannot draw become ASCII, or `?` when there is +// no spelling for them. +func compactASCII(hint string) string { + return strings.Map(func(r rune) rune { + if r <= unicode.MaxASCII || unicode.IsLetter(r) || unicode.IsDigit(r) || unicode.IsSpace(r) || unicode.IsMark(r) { + return r + } + return '?' + }, compactASCIIPunctuation.Replace(hint)) +} + +// divider is the compaction mark: one dim line with the fact in it, because a // conversation that silently lost its middle is a conversation the person // cannot reason about. +// +// IT IS A LINE AND NOT A RULE. It was a full-width rule with the fact centred +// in it, the loudest shape on the page for the surface's own housekeeping, and +// the design language is dim telemetry with no borders. It now wears the +// note lane's lead, so it reads as what it is — something codeaf did, said +// once, quietly — and the mark tells it apart from the notes around it. func (a *app) divider(hint string, width int) string { - label := " ⚭ " + hint + " " - rest := width - ansi.StringWidth(label) - 2 - if rest < 0 { - return a.pal.dim(ansi.Truncate("──"+label, width, glyphMore)) + lead, more := "· ", glyphMore + if a.linear || a.pal.ascii { + lead = "- " + more = ">" + hint = compactASCII(hint) } - left := rest / 2 - return a.pal.dim(strings.Repeat("─", left+2) + label + strings.Repeat("─", rest-left)) + return a.pal.dim(ansi.Truncate(lead+a.icon(tokens.GCompacted)+" "+hint, width, more)) } // compactRow draws one compaction pass, in the two shapes it has. @@ -1889,9 +1912,10 @@ func (a *app) divider(hint string, width int) string { // violet is not available to it — that hue means a person is being asked // something, and nobody is being asked anything here.) // -// SETTLED, it is the rule it always was, with what it cost in time: +// SETTLED, it is the one dim line [app.divider] draws, with what it cost in +// time, and it stays standing when the turn's work folds (workfold.go): // -// ───── ⚭ compacted from ~84k tokens · took 6s ───── +// · ⚭ compacted · summarized 4 messages · ~31k → ~13k tokens · took 6s // // The duration is dropped under a second, by the same law the tool clock uses // ([countUpWord]'s floor): "took 0s" is a column read for nothing. diff --git a/internal/tui3/tui3_test.go b/internal/tui3/tui3_test.go index 04dab46dfc..3b6cfcb853 100644 --- a/internal/tui3/tui3_test.go +++ b/internal/tui3/tui3_test.go @@ -1203,17 +1203,20 @@ func TestCompactionDrawsADivider(t *testing.T) { a := newTestApp(agent) runTurn(t, a, agent, "keep going") - // The compaction mark is machinery, so a turn that settles on an answer - // tucks it away with the rest of the work (workfold.go). Nobody wants to be - // told the context was squeezed while they are reading the reply; they want - // it when they go looking for why, which is what ctrl+e is for. + // A PASS THAT ONLY STUBBED AND FOLDED IS MACHINERY, so a turn that settles + // on an answer tucks it away with the rest of the work (workfold.go), and + // ctrl+e brings it back — one quiet line, not a rule across the page. Only + // a pass that wrote a summary stands ([TestACompactionMidTurnStaysVisibleWhenTheWorkFolds]). if folded := plain(frame(a)); strings.Contains(folded, "compacted from") { - t.Fatalf("a settled turn still shows the compaction mark:\n%s", folded) + t.Fatalf("a settled turn still shows a free pass's mark:\n%s", folded) } drive(t, a, key("ctrl+e")) got := plain(frame(a)) - if !strings.Contains(got, "⚭ compacted from ~84k tokens") || !strings.Contains(got, "──") { - t.Fatalf("the compaction divider is missing:\n%s", got) + if !strings.Contains(got, "· ⚭ compacted from ~84k tokens") { + t.Fatalf("the opened work does not carry the compaction line:\n%s", got) + } + if row := findRow(t, a, "compacted from ~84k tokens"); strings.Contains(row, "──") { + t.Fatalf("the compaction mark is still a rule: %q", row) } } diff --git a/internal/tui3/workfold.go b/internal/tui3/workfold.go index 9dd86c46c0..6f60e9c3dd 100644 --- a/internal/tui3/workfold.go +++ b/internal/tui3/workfold.go @@ -1,7 +1,7 @@ package tui3 import ( - "github.com/Agent-Field/codeaf/internal/tui2/tokens" + "sort" "strings" "time" @@ -9,6 +9,7 @@ import ( "github.com/Agent-Field/codeaf/internal/config" "github.com/Agent-Field/codeaf/internal/session" + "github.com/Agent-Field/codeaf/internal/tui2/tokens" ) // workfold is render-time structure. Nothing here is journaled: replaying the @@ -272,7 +273,7 @@ func deriveWorkfolds(es []entry, runningTurn int) map[int]workfold { // machine the router refuses — and a chip that hid it left them with a // pin that disappeared and no sentence anywhere saying why. blocked, stopped := false, false - var asks []int + var asks, compactions []int for i := lo; i < hi; i++ { if es[i].kind == entryAssistant && strings.TrimSpace(es[i].text) != "" { answer = i @@ -290,6 +291,16 @@ func deriveWorkfolds(es []entry, runningTurn int) map[int]workfold { if es[i].kind == entryTool && ((es[i].status != toolOK && es[i].status != toolFailed && !es[i].cut) || es[i].decision != "") { asks = append(asks, i) } + // A FINISHED SUMMARY STANDS LIKE AN ASK. A pass that summarized + // rewrote the person's own words in the model's copy of the + // conversation, and it is rare, so its line stays when the turn's + // work folds. A pass that only stubbed and folded is machinery like + // any other step and folds with it (#1627's intent): on a small + // window those run almost every step, and a standing line for each + // drew five marks between five chips on one turn (review of #1658). + if es[i].kind == entryCompact && !es[i].ended.IsZero() && es[i].summarized { + compactions = append(compactions, i) + } if es[i].kind == entryNote && (es[i].told || strings.HasPrefix(es[i].text, "cancel")) { asks = append(asks, i) } @@ -319,6 +330,13 @@ func deriveWorkfolds(es []entry, runningTurn int) map[int]workfold { } } } + // A SUMMARY STANDS WHEREVER IT RAN. Before the answer it splits the + // fold; after it — the end-of-turn check, the ordinary case — it would + // otherwise be taken into the turn's disclosure as bookkeeping + // ([housekeepingFold]), and a person reading back would find no sign + // that their words were summarized. + asks = append(asks, compactions...) + sort.Ints(asks) // AN ASK STANDS, AND THE WORK BEFORE IT STILL FOLDS. A task proposal, a // sign-in or a standing card is a thing the work could not decide alone, // so no chip may cover it. It used to keep the WHOLE turn open instead: