From 7ec93a0cc65e5c41c8cd9554a6c15a4b278ec6cc Mon Sep 17 00:00:00 2001 From: Markus Neusinger <2921697+MarkusNeusinger@users.noreply.github.com> Date: Sat, 10 Oct 2026 00:51:29 +0200 Subject: [PATCH 1/3] feat(api): a chat turn may wait the full queue maximum anyplot-api's Cloud Run request timeout rises from 600 to 900 s and the BFF's AGENT_TURN_MAX_S from 590 to 890 s, so a turn at the back of a full run queue (600 s of waiting plus the 180 s run) ends with its own error and done. The design doc records the owner's decisions of 2026-10-10: the capacity formula stays, and one run or toggle in flight per user stays. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01GdMSqLR5ww4ji74EUSmk9M --- .env.example | 4 ++-- api/cloudbuild.yaml | 5 ++++- changelog.d/agents-turn-cap.md | 9 +++++++++ core/config.py | 4 ++-- docs/concepts/agent-network.md | 13 +++++++------ docs/reference/api.md | 6 +++--- 6 files changed, 27 insertions(+), 14 deletions(-) create mode 100644 changelog.d/agents-turn-cap.md diff --git a/.env.example b/.env.example index 660ad392ec..717216a09c 100644 --- a/.env.example +++ b/.env.example @@ -78,8 +78,8 @@ PORT=8000 # the agents service's run queue. # AGENT_REQUEST_TIMEOUT_S=190 # Hard cap in seconds on one chat turn, queue wait included; keep it below -# anyplot-api's Cloud Run --timeout (600). -# AGENT_TURN_MAX_S=590 +# anyplot-api's Cloud Run --timeout (900). +# AGENT_TURN_MAX_S=890 # ============================================================================ # AI Services (optional) diff --git a/api/cloudbuild.yaml b/api/cloudbuild.yaml index 6d5dfa9428..f1284b1057 100644 --- a/api/cloudbuild.yaml +++ b/api/cloudbuild.yaml @@ -118,7 +118,10 @@ steps: # two-slot semaphore (api/routers/og_images.py), so 40 in-flight requests # cannot become 40 concurrent PIL renders. - "--concurrency=40" - - "--timeout=600" + # 900 s: an agent chat turn may wait up to 600 s in the agents service's + # run queue and then run for 180 s; the BFF caps a turn at + # AGENT_TURN_MAX_S (890 s), which has to stay below this timeout. + - "--timeout=900" # Deploy WITHOUT routing traffic: the smoke step probes this revision on # its tag URL first, and only the promote step below shifts traffic — so a # bad image never serves users, not even between deploy and a failing diff --git a/changelog.d/agents-turn-cap.md b/changelog.d/agents-turn-cap.md new file mode 100644 index 0000000000..815ffbc59b --- /dev/null +++ b/changelog.d/agents-turn-cap.md @@ -0,0 +1,9 @@ +### Changed + +- **A chat turn may wait the full queue maximum.** anyplot-api's Cloud Run + request timeout rises from 600 to 900 seconds and the BFF's turn cap + `AGENT_TURN_MAX_S` from 590 to 890 seconds, so a turn at the back of a full + run queue (600 seconds of waiting plus the 180-second run) ends with its own + `error` and `done` instead of being cut at about 385 seconds. The design + document records the owner's decisions of 2026-10-10: the queue's capacity + formula stays, and one run or theme toggle in flight per user stays. diff --git a/core/config.py b/core/config.py index 6b34a2366f..e936dcdaca 100644 --- a/core/config.py +++ b/core/config.py @@ -286,9 +286,9 @@ def _parse_admin_allowed_emails(cls, value: Any) -> Any: above the agents service's own 180 s request deadline, so its `deadline` error arrives before the BFF gives up on the stream.""" - agent_turn_max_s: int = 590 + agent_turn_max_s: int = 890 """Hard wall-clock cap in seconds on one relayed chat turn, queue wait - included. Keep it below anyplot-api's Cloud Run `--timeout` (600 s in + included. Keep it below anyplot-api's Cloud Run `--timeout` (900 s in `api/cloudbuild.yaml`), so Cloud Run never cuts a stream without `error` and `done`. A turn still queued when less than `agent_request_timeout_s` plus the queue's 15 s heartbeat is left ends with `capacity` and leaves the diff --git a/docs/concepts/agent-network.md b/docs/concepts/agent-network.md index 6c15dd8536..31511b3282 100644 --- a/docs/concepts/agent-network.md +++ b/docs/concepts/agent-network.md @@ -143,7 +143,7 @@ The owner decided the queue, the rate and the serial renders on 2026-10-09, afte | Request deadline | 180 s through `abort_signal` plus task cancellation on disconnect, counted from the moment the run leaves the queue | the `/messages` route | | Runs in flight per instance | 1 (`AGENT_RUN_CONCURRENCY`) | the run queue | | Run starts | fewer than `AGENT_RUNS_PER_MINUTE` (1) in the last 60 s, a sliding window of start times | the run queue | -| Wait in the queue | at most `AGENT_QUEUE_MAX_WAIT_S` (600 s); a run that waited that long ends with `error{code:"capacity"}`, unless its turn comes at that very moment. Through the BFF, a turn waits at most about 385 s until anyplot-api's request timeout is raised (see [Risks](#risks-and-mitigations)) | the run queue; `AGENT_TURN_MAX_S` in the BFF | +| Wait in the queue | at most `AGENT_QUEUE_MAX_WAIT_S` (600 s); a run that waited that long ends with `error{code:"capacity"}`, unless its turn comes at that very moment. Through the BFF, a whole turn ends by `AGENT_TURN_MAX_S` (890 s), below anyplot-api's 900 s request timeout, so the full wait plus the 180 s run fits | the run queue; `AGENT_TURN_MAX_S` in the BFF | | Queue length | rate x maximum wait / 60, so 10 entries at the defaults; a new entry past that is refused with `503 capacity` before the stream starts. The formula assumes runs within the 60 s rate window, so admission promises a place, not a start | the run queue | | Runs per user | one queued or running run, or one theme toggle in flight, in any session; anything more gets `409 run_active` | the run registry | | Daily budget | a user already over the daily token budget gets the `budget` refusal at once and never takes a place in the queue | the `/messages` route | @@ -272,7 +272,7 @@ Testing a newer model version, the other provider's arm, a different judge model - **Environment on anyplot-agents:** `ENVIRONMENT=production`, `GOOGLE_CLOUD_PROJECT=anyplot`, `GOOGLE_GENAI_USE_ENTERPRISE=TRUE`, `GOOGLE_CLOUD_LOCATION=eu`, `AGENT_LOCATION=eu` (never europe-west4), `AGENT_PROVIDER=anthropic-vertex`, `AGENT_MODEL=claude-haiku-5-5`, `AGENT_JUDGE_MODEL=claude-haiku-5-5` (the Gemini arm: `AGENT_PROVIDER=gemini`, `AGENT_MODEL=gemini-3.8-flash`, `AGENT_JUDGE_MODEL=gemini-3.5-flash-lite`), `AGENT_LIBRARIES=matplotlib,seaborn`, `AGENT_RENDERER=sandbox`, `AGENT_MAX_LLM_CALLS=12`, `ADK_MAX_LLM_CALLS=20`, `AGENT_REQUEST_TOKEN_BUDGET=80000`, `AGENT_DAILY_TOKEN_BUDGET=1000000`, `AGENT_DAILY_PIPELINE_RUNS=40`, `AGENT_GLOBAL_DAILY_TOKEN_BUDGET=3000000`, `AGENT_RUN_CONCURRENCY=1`, `AGENT_RUNS_PER_MINUTE=1`, `AGENT_QUEUE_MAX_WAIT_S=600`, `AGENT_RENDER_CONCURRENCY=1`, `AGENT_RENDER_TIMEOUT_S=60`, `AGENT_REQUEST_DEADLINE_S=180`, `AGENT_SOFT_DEADLINE_S=140`, `AGENT_ALLOWED_CALLERS=`, `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false`. On anyplot-api: `AGENT_ENABLED=false` (ships dark; the routes answer 404), `AGENT_SERVICE_URL`, and `AGENT_USER_ID_KEY` from Secret Manager (the BFF also answers 404 while the key is unset, so a deploy never breaks). On the app: `VITE_ENABLE_AGENT_CHAT`, a build-time flag that tree-shakes the chunk. - **Sessions and artifacts.** Phase 1 uses `InMemorySessionService`, `InMemoryArtifactService` and in-memory dataset and render stores; they are consistent because `max-instances=1`, and scale-to-zero shows "session expired". Phase 2 moves to `DatabaseSessionService` in a separate database `anyplot_agents` on `anyplot-db` with its own user and an hourly purge of sessions older than 24 h, because `alembic/env.py` has no `include_object` filter and ADK's `create_all` tables would read as drift; artifacts go to `GcsArtifactService` on a private EU bucket with a 1-day lifecycle, never `anyplot-images`. - **BFF** in `api/routers/agent.py`: `APIRouter(prefix="/debug/agent", dependencies=[Depends(require_admin)])`. A small refactor `require_admin_identity()` returns `AdminIdentity(email|None, via)` and `require_admin` wraps it unchanged. `user_id = "adm_" + HMAC(key, email or "token")[:16]`. POST requests need `Content-Type: application/json`, `X-Anyplot-Client: agent-chat/1` and an allowed Origin. ID tokens come from `google.oauth2.id_token.fetch_id_token` (skipped for localhost). The messages route is an async generator declared with `response_class=EventSourceResponse` (which is what gives FastAPI's automatic 15 s pings; Cloudflare returns 524 after 125 s), reads upstream with httpx `aiter_lines()`, assembles complete events, re-validates each event type against the `anyplot/1` allowlist, and yields `ServerSentEvent(event=..., raw_data=...)`; on upstream failure it emits `error{code:"upstream"}`. Routes mirror `/v1` plus `GET eligibility`, `POST sessions/{sid}/feedback`, and the triage routes `GET /debug/agent/cases`, `GET /debug/agent/cases/{id}` and `PATCH /debug/agent/cases/{id}`. The deploy smoke test expects 401 on `/debug/agent/status`. -- **SSE protocol `anyplot/1`**, translated from ADK events and never forwarded raw: `ready{v, run_id}`, sent before the run waits in the queue; `status{step:"queued", position, waiting}` while it waits, at once, on every change and every 15 s unchanged (`position` 1 runs next, `waiting` counts every queued entry, this one included); `status{step, attempt}` for the pipeline's steps; `message{text}` (root final text only), `plot{PlotResult}`, `refusal{code, text}`, `error{code, ref}` with codes `capacity` (also for a run that waited the queue's maximum), `deadline`, `guard_unavailable`, `upstream` and `internal`, and `done{llm_calls, tokens}`. The translator tolerates the ADK 2.x `node_info` and `output` fields. A user already over the daily budget gets `ready`, `refusal{code:"budget"}` and `done` without waiting in the queue. Because queued time does not count toward the agents service's deadline, the BFF restarts its own turn budget (`AGENT_REQUEST_TIMEOUT_S`) on every queued status, with the queue's 15 s heartbeat on top because the run may start that long before its first event, and on the first event after the wait. Nothing moves a turn past `AGENT_TURN_MAX_S` (590 s), which stays below anyplot-api's `--timeout=600`: a turn still queued when its run could no longer finish inside that cap ends with `error{code:"capacity"}`, and closing the upstream stream takes it out of the queue before it spends a token. +- **SSE protocol `anyplot/1`**, translated from ADK events and never forwarded raw: `ready{v, run_id}`, sent before the run waits in the queue; `status{step:"queued", position, waiting}` while it waits, at once, on every change and every 15 s unchanged (`position` 1 runs next, `waiting` counts every queued entry, this one included); `status{step, attempt}` for the pipeline's steps; `message{text}` (root final text only), `plot{PlotResult}`, `refusal{code, text}`, `error{code, ref}` with codes `capacity` (also for a run that waited the queue's maximum), `deadline`, `guard_unavailable`, `upstream` and `internal`, and `done{llm_calls, tokens}`. The translator tolerates the ADK 2.x `node_info` and `output` fields. A user already over the daily budget gets `ready`, `refusal{code:"budget"}` and `done` without waiting in the queue. Because queued time does not count toward the agents service's deadline, the BFF restarts its own turn budget (`AGENT_REQUEST_TIMEOUT_S`) on every queued status, with the queue's 15 s heartbeat on top because the run may start that long before its first event, and on the first event after the wait. Nothing moves a turn past `AGENT_TURN_MAX_S` (890 s), which stays below anyplot-api's `--timeout=900`: a turn still queued when its run could no longer finish inside that cap ends with `error{code:"capacity"}`, and closing the upstream stream takes it out of the queue before it spends a token. - **Frontend.** A lazy `app/src/pages/AgentChatPage.tsx` at `debug/agent` with a data panel (textarea with a 200 KB counter, preview table, binding dropdowns), the chat thread, a progress timeline and a Stop button (`AbortController` plus the cancel route). The **result card** (`app/src/sections/agent-chat/ResultCard.tsx`) reuses the plot page's overlay actions: the rendered plot shown inline as a blob URL with a light and dark switch that renders the other theme on demand through the theme toggle route, **Copy image** (Clipboard API `navigator.clipboard.write([new ClipboardItem({"image/png": blob})])`, falling back to download where unsupported), **Download PNG** for every rendered theme and **Open full size**; the adapted code in `CodeHighlighter` with **Copy code** (`useCopyCode`, one click), **Download plot.py** and **Download data.csv** (the pair runs unchanged); the change list and residual notes; the quick-feedback control; and a composer for refinements. Earlier versions stay reachable in the thread, each with its own image and code. The same card is reused unchanged when the feature goes public. `app/src/lib/sse.ts` parses the stream over `fetchWithAuth(...).body.getReader()`. The `.adapt()` button is the fourth overlay button in `app/src/sections/spec-detail/SpecDetailView.tsx`, wired through `onUseWithMyData` from `SpecPage.tsx`, and renders only when `CONFIG.features.agentChat && (CONFIG.isDev || adminHint) && eligible`, where `adminHint` is a localStorage flag that `DebugPage` sets after `/debug/status` succeeds and `eligible` comes from the eligibility route. It does a full navigation so Cloudflare Access can intercept, and public pages never probe `/api/debug/*`. - **Analytics** (enum properties only, documented in [Plausible](../reference/plausible.md) when implemented): the pageview `/debug/agent`, `agent_open{library,source}`, `agent_data_parsed{status,size_bucket}`, `agent_plot_rendered{library,status,repaired}`, `agent_guardrail_block{reason}`, `agent_result_feedback{reaction,include_data}`, and `copy_code{page:'agent_chat',method:'agent'}`. @@ -400,8 +400,9 @@ Defaults apply until the owner decides otherwise. - Retention periods for the legal page (phase 2). - Scale-to-zero "session expired" versus `min-instances=1`: accept in phase 1. - A named fallback model on `eu` for capacity failover, which deviates from one pinned model everywhere: none. -- The request timeout for a queued turn: a run at the back of a full queue needs up to 780 s (600 s in the queue, 180 s to run), more than anyplot-api's `--timeout=600`. Until that changes, the BFF caps a turn at `AGENT_TURN_MAX_S` (590 s) and ends a turn still queued after about 385 s with `capacity`, so the full 600 s wait is not reachable through the BFF. To use it, raise anyplot-api's timeout to about 900 s and `AGENT_TURN_MAX_S` to about 890 s; or lower `AGENT_QUEUE_MAX_WAIT_S` to about 385 s so the queue's own promise matches the BFF: open. -- The queue length: the owner's formula (rate x maximum wait / 60) assumes runs within the 60 s rate window, so with longer runs the back of a full queue waits the full maximum and gets `capacity`. Admitting by an estimated start time (position times the larger of 60 s per start and the recent run time, divided by the concurrency) would keep the promise: open; the formula stays until then. +- The request timeout for a queued turn: a run at the back of a full queue needs up to 780 s (600 s in the queue, 180 s to run). Decided on 2026-10-10: anyplot-api runs with `--timeout=900` and the BFF caps a turn at `AGENT_TURN_MAX_S` (890 s), so the full wait is reachable through the BFF. +- The queue length: the owner's formula (rate x maximum wait / 60) assumes runs within the 60 s rate window, so with longer runs the back of a full queue waits the full maximum and gets `capacity`. Decided on 2026-10-10: the formula stays; admitting by an estimated start time (position times the larger of 60 s per start and the recent run time, divided by the concurrency) is the fallback if the live service shows many expiries. +- One run or toggle in flight per user, across sessions (stricter than refusing a toggle only while that session runs): decided on 2026-10-10, kept. - Whether the "Create plot" action carries the site theme, so a dark-mode visitor's first render is already dark instead of a light render plus a toggle: not yet; the action renders light. - Feedback-case retention and data inclusion: 180 days in the private bucket; `include_data` on by default for admins and off by default with explicit consent for real users. - Premium positioning: [Vision](vision.md) lists "Try with your data" as premium; the licensing row in the decisions table settles that the code stays MIT regardless. @@ -417,8 +418,8 @@ Defaults apply until the owner decides otherwise. | ADK ships weekly with feature-flag drift | An exact pin to 2.11.0; re-check `FUNCTION_TOOL_ARG_VALIDATION` and `JSON_SCHEMA_FOR_FUNC_DECL` on upgrade; plain-code modules without ADK imports; the harness report stamps the ADK version | | Caller trust: anyplot-api runs as the shared compute service account | Acceptable while admin-only; the IAM-forwarded claims check; a dedicated service account before public use | | Deadline overrun (2 renders of one theme times 60 s plus LLM calls) | A 140 s soft deadline inside the pipeline (skip the repair when short, clamp the render timeout), a `try/finally` that always yields a `PlotResult`, the 180 s `abort_signal` backstop | -| Long SSE streams hold anyplot-api concurrency slots on its single instance | The 180 s deadline, the 590 s turn cap, one run per user, and a queue of at most 10 waiting streams; the route moves out of the API before public use | -| A queued turn outlasts a request timeout on its path: the 600 s maximum wait plus the 180 s deadline is 780 s, while anyplot-api runs with `--timeout=600` | The BFF restarts its turn budget while the run waits but never past `AGENT_TURN_MAX_S` (590 s): a turn still queued after about 385 s ends with `error{code:"capacity"}` and leaves the queue before it spends a token, so Cloud Run never cuts a stream silently. Using the full wait needs anyplot-api's timeout raised (an owner decision) | +| Long SSE streams hold anyplot-api concurrency slots on its single instance | The 180 s deadline, the 890 s turn cap, one run per user, and a queue of at most 10 waiting streams; the route moves out of the API before public use | +| A queued turn outlasts a request timeout on its path: the 600 s maximum wait plus the 180 s deadline is 780 s | anyplot-api runs with `--timeout=900`; the BFF restarts its turn budget while the run waits but never past `AGENT_TURN_MAX_S` (890 s): a turn still queued when its run could no longer finish inside the cap ends with `error{code:"capacity"}` and leaves the queue before it spends a token, so Cloud Run never cuts a stream silently | | Out of memory with more than one sandbox (spikes S and S2) | One run in flight; serial renders by construction, because `SerialRenderer` in `Services.backend` puts every backend and every route (the theme toggle and `adk web` included) behind the same slots, so a backend cannot forget its semaphore | | Theme toggles crowd out paid runs (renders that cost no tokens but hold the one render slot) | A freed render slot goes to a waiting pipeline render first; one run or toggle in flight per user, and a toggle and a turn refuse each other in a session; a toggle waits for a slot at most 120 s; a theme that failed the host gates is rendered at most twice | | Cloudflare 524 or Worker buffering | The generator route with automatic 15 s pings and first bytes sent immediately, verified in the phase-1 smoke test; asynchronous polling for slow runtimes later | diff --git a/docs/reference/api.md b/docs/reference/api.md index 93d2ddd98e..173b9bd383 100644 --- a/docs/reference/api.md +++ b/docs/reference/api.md @@ -558,8 +558,8 @@ sends `error {"code": "upstream"}` and then `done {}`. Time in the run queue does not count toward `AGENT_REQUEST_TIMEOUT_S`: every `queued` status starts the budget again with 15 seconds on top, because the run can start up to one `queued` status before its first event, and the first event after the wait -starts it again. No turn runs past `AGENT_TURN_MAX_S` (590 seconds), which -stays below anyplot-api's own request timeout of 600 seconds: a turn that is +starts it again. No turn runs past `AGENT_TURN_MAX_S` (890 seconds), which +stays below anyplot-api's own request timeout of 900 seconds: a turn that is still queued when its run could no longer finish inside that cap ends with `error {"code": "capacity"}` and `done {}`, and leaves the queue before it spends a token. At the defaults, a turn waits at most about 385 seconds @@ -595,7 +595,7 @@ body is never echoed. Two cases answer `502` instead: | `AGENT_SERVICE_URL` | Unset | Base URL of anyplot-agents, without `/v1`; also the ID-token audience | | `AGENT_USER_ID_KEY` | Unset | HMAC key for the user id; from Secret Manager in production | | `AGENT_REQUEST_TIMEOUT_S` | `190` | Upstream timeout, and the cap on one chat stream outside the run queue; above the agents service's 180-second deadline | -| `AGENT_TURN_MAX_S` | `590` | Hard cap on one chat turn, queue wait included; keep it below anyplot-api's Cloud Run `--timeout` (600 seconds in `api/cloudbuild.yaml`) | +| `AGENT_TURN_MAX_S` | `890` | Hard cap on one chat turn, queue wait included; keep it below anyplot-api's Cloud Run `--timeout` (900 seconds in `api/cloudbuild.yaml`) | In production, `AGENT_ENABLED` and `AGENT_SERVICE_URL` come from the `_AGENT_ENABLED` and `_AGENT_SERVICE_URL` substitutions in `api/cloudbuild.yaml`. From 17dd4ad21e53bcabc6312c9d44f298071ab1fa11 Mon Sep 17 00:00:00 2001 From: Markus Neusinger <2921697+MarkusNeusinger@users.noreply.github.com> Date: Sat, 10 Oct 2026 00:52:02 +0200 Subject: [PATCH 2/3] chore(changelog): reference #12113 in the turn-cap fragment Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01GdMSqLR5ww4ji74EUSmk9M --- changelog.d/agents-turn-cap.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/changelog.d/agents-turn-cap.md b/changelog.d/agents-turn-cap.md index 815ffbc59b..353ae95db6 100644 --- a/changelog.d/agents-turn-cap.md +++ b/changelog.d/agents-turn-cap.md @@ -6,4 +6,4 @@ run queue (600 seconds of waiting plus the 180-second run) ends with its own `error` and `done` instead of being cut at about 385 seconds. The design document records the owner's decisions of 2026-10-10: the queue's capacity - formula stays, and one run or theme toggle in flight per user stays. + formula stays, and one run or theme toggle in flight per user stays. (#12113) From de7430ac2ae9fd7d537e3afb917a2983f7d2fff9 Mon Sep 17 00:00:00 2001 From: Markus Neusinger <2921697+MarkusNeusinger@users.noreply.github.com> Date: Sat, 10 Oct 2026 00:54:55 +0200 Subject: [PATCH 3/3] docs(agents): the 890 s cap reaches the full queue wait everywhere The config docstring, the agents README, the API reference and the changelog fragment no longer describe the old 385-second effective limit. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01GdMSqLR5ww4ji74EUSmk9M --- agents/README.md | 2 +- changelog.d/agents-turn-cap.md | 7 ++++--- core/config.py | 4 ++-- docs/reference/api.md | 5 +++-- 4 files changed, 10 insertions(+), 8 deletions(-) diff --git a/agents/README.md b/agents/README.md index 08f7474877..3d5eb2a553 100644 --- a/agents/README.md +++ b/agents/README.md @@ -55,7 +55,7 @@ Three rules hold for everything here: | `AGENT_RENDER_CONCURRENCY` | `1` | Theme renders at the same time, for every backend (`render/serial.py`); serial, because one 4 GiB instance holds one sandbox safely (spikes S and S2) | | `AGENT_RUN_CONCURRENCY` | `1` | Pipeline runs (whole `/messages` turns) in flight; the run queue holds the rest | | `AGENT_RUNS_PER_MINUTE` | `1` | Runs that may start within any 60 seconds (a sliding window) | -| `AGENT_QUEUE_MAX_WAIT_S` | `600` | Longest wait in the run queue, after which the run ends with `capacity`; the queue holds rate x wait / 60 entries (10) and answers `503 capacity` beyond that. Through the BFF a turn waits at most about 385 s until anyplot-api's request timeout is raised (`AGENT_TURN_MAX_S` in `docs/reference/api.md`) | +| `AGENT_QUEUE_MAX_WAIT_S` | `600` | Longest wait in the run queue, after which the run ends with `capacity`; the queue holds rate x wait / 60 entries (10) and answers `503 capacity` beyond that. Through the BFF the whole turn, this wait included, ends by `AGENT_TURN_MAX_S` (890 s, see `docs/reference/api.md`), which the full wait plus the 180 s run fits into | | `AGENT_MAX_LLM_CALLS` | `12` | LLM calls per request | | `AGENT_REQUEST_TOKEN_BUDGET` | `80000` | Tokens per request | | `AGENT_DAILY_TOKEN_BUDGET` | `1000000` | Tokens per user and day | diff --git a/changelog.d/agents-turn-cap.md b/changelog.d/agents-turn-cap.md index 353ae95db6..7bf85f3b83 100644 --- a/changelog.d/agents-turn-cap.md +++ b/changelog.d/agents-turn-cap.md @@ -2,8 +2,9 @@ - **A chat turn may wait the full queue maximum.** anyplot-api's Cloud Run request timeout rises from 600 to 900 seconds and the BFF's turn cap - `AGENT_TURN_MAX_S` from 590 to 890 seconds, so a turn at the back of a full - run queue (600 seconds of waiting plus the 180-second run) ends with its own - `error` and `done` instead of being cut at about 385 seconds. The design + `AGENT_TURN_MAX_S` from 590 to 890 seconds. The old cap already ended a + turn cleanly, but only after about 385 seconds of waiting; now a turn at + the back of a full run queue (600 seconds of waiting plus the 180-second + run) reaches its run instead of ending with `capacity`. The design document records the owner's decisions of 2026-10-10: the queue's capacity formula stays, and one run or theme toggle in flight per user stays. (#12113) diff --git a/core/config.py b/core/config.py index e936dcdaca..3b3f2b1a3a 100644 --- a/core/config.py +++ b/core/config.py @@ -292,8 +292,8 @@ def _parse_admin_allowed_emails(cls, value: Any) -> Any: `api/cloudbuild.yaml`), so Cloud Run never cuts a stream without `error` and `done`. A turn still queued when less than `agent_request_timeout_s` plus the queue's 15 s heartbeat is left ends with `capacity` and leaves the - queue before it spends a token, so at the defaults a turn waits at most - about 385 s through the BFF, not the agents service's 600 s maximum.""" + queue before it spends a token; at the defaults the agents service's full + 600 s queue maximum plus its 180 s run fit inside the cap.""" # ============================================================================= # CORS diff --git a/docs/reference/api.md b/docs/reference/api.md index 173b9bd383..fd65c7ccc5 100644 --- a/docs/reference/api.md +++ b/docs/reference/api.md @@ -562,8 +562,9 @@ starts it again. No turn runs past `AGENT_TURN_MAX_S` (890 seconds), which stays below anyplot-api's own request timeout of 900 seconds: a turn that is still queued when its run could no longer finish inside that cap ends with `error {"code": "capacity"}` and `done {}`, and leaves the queue before it -spends a token. At the defaults, a turn waits at most about 385 seconds -through the BFF. Errors that happen before the stream starts, such as +spends a token. At the defaults, the agents service's full 600-second queue +maximum plus the 180-second run fit inside the cap, so the BFF never cuts a +wait short. Errors that happen before the stream starts, such as `413 too_long`, an upstream `409 run_active`, or `503 capacity` from a full run queue, arrive as HTTP statuses instead.