From 230af591636aea1b82e8ee85d3e2d169e78d0295 Mon Sep 17 00:00:00 2001 From: DABH Date: Fri, 11 Sep 2026 01:15:06 -0500 Subject: [PATCH 01/11] Add OpenRouter integration page for Python Documents calling OpenRouter from Activities with Temporal-owned retries, error classification, response caching for free retries, fan-out, and a budget gate that pauses on a soft budget or a 402. Adds the sidebar entry and registry entries for Python and TypeScript under a new Model gateway tag. Snippets are rendered from temporalio/samples-python#TBD. --- .../python/integrations/openrouter.mdx | 392 ++++++++++++++++++ sidebars.js | 1 + .../IntegrationsGrid/integrations-data.json | 18 + 3 files changed, 411 insertions(+) create mode 100644 docs/develop/python/integrations/openrouter.mdx diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx new file mode 100644 index 0000000000..90cadf32e1 --- /dev/null +++ b/docs/develop/python/integrations/openrouter.mdx @@ -0,0 +1,392 @@ +--- +id: openrouter +title: OpenRouter integration +sidebar_label: OpenRouter +toc_max_heading_level: 2 +tags: + - OpenRouter + - Python SDK + - Temporal SDKs +description: + Call OpenRouter from Temporal Activities with durable retries, cost tracking, and budgets, using the Temporal Python + SDK. +--- + +[OpenRouter](https://openrouter.ai/) is a model gateway: one OpenAI-compatible API and one API key in front of hundreds +of models from many providers. It picks the provider and model for each request, falls back between them, and reports +what every response cost. + +Temporal handles everything around those calls. Each call runs as an Activity, so it gets retries with backoff, a +timeout, and a durable record in Event History of what was called and what it cost. The Workflow around the +Activities can fan out over a batch with bounded concurrency, survive a Worker crash without re-running finished +calls, and pause for hours until a person acts. + +The division of labor is simple. OpenRouter decides *which provider and model* serve a request, in milliseconds. +Temporal decides *what happens over time*: waiting out a rate limit, surviving a crash, parking until someone raises a +budget, and keeping the audit trail. + +This integration is a sample pattern rather than a plugin: OpenRouter needs nothing inside Workflow code. Code snippets +in this guide are taken from the [OpenRouter samples](https://github.com/temporalio/samples-python/tree/main/openrouter). +Refer to the samples for the complete code, and to the +[TypeScript sample](https://github.com/temporalio/samples-typescript/tree/main/openrouter) for the same pattern in +TypeScript. + +## Prerequisites + +- This guide assumes you are already familiar with OpenRouter's + [chat completions API](https://openrouter.ai/docs/quickstart). +- If you are new to Temporal, we recommend reading [Understanding Temporal](/evaluate/understanding-temporal) or taking + the [Temporal 101](https://learn.temporal.io/courses/temporal_101/) course. +- Set up your local development environment by following + [Set up your local development environment](/develop/python/set-up-your-local-python) and leave the Temporal + development server running. +- Create an [OpenRouter API key](https://openrouter.ai/settings/keys) and export it as `OPENROUTER_API_KEY` in the + Worker's environment. The key stays in the Worker process; it is never part of Workflow input or Event History. + +## Install + +The samples call OpenRouter with the `openai` package pointed at OpenRouter's base URL, which is the setup OpenRouter +documents for OpenAI-compatible clients: + +```bash +pip install temporalio openai +``` + +OpenRouter's official [`openrouter`](https://pypi.org/project/openrouter/) package works the same way. If you use it, +construct the client with `retry_config=RetryConfig("none", ...)`. By default it retries 5xx and connection errors for +up to an hour, which hides attempts from Temporal. + +## Call OpenRouter from an Activity + +Build one client for the Worker's lifetime with client-side retries turned off, and pass it to the Activity class. The +Activity makes exactly one HTTP call per attempt, so every attempt is visible in Event History and the Activity's Retry +Policy is the only retry policy in play. + +```python +from openai import AsyncOpenAI + +client = AsyncOpenAI( + base_url="https://openrouter.ai/api/v1", + api_key=os.environ["OPENROUTER_API_KEY"], + max_retries=0, + timeout=60.0, +) +``` + + +[openrouter/activities.py](https://github.com/temporalio/samples-python/blob/main/openrouter/activities.py) +```py +@activity.defn +async def call_openrouter(self, request: OpenRouterRequest) -> OpenRouterResult: + """One chat completion. One HTTP call per attempt; Temporal retries.""" + # Heartbeat so a killed Worker is noticed after heartbeat_timeout + # rather than after the full start_to_close_timeout. + heartbeat_timeout = activity.info().heartbeat_timeout + heartbeat_task = ( + asyncio.create_task(_heartbeat_forever(heartbeat_timeout / 2)) + if heartbeat_timeout + else None + ) + try: + return await self._send(request) + finally: + if heartbeat_task: + heartbeat_task.cancel() + +async def _send(self, request: OpenRouterRequest) -> OpenRouterResult: + extra_body: dict[str, Any] = {} + if request.fallback_models: + # OpenRouter tries these in order within the same request. + extra_body["models"] = request.fallback_models + elif request.model == "openrouter/auto": + extra_body["plugins"] = [ + {"id": "auto-router", "cost_tier": request.cost_tier} + ] + model = request.fallback_models[0] if request.fallback_models else request.model + + try: + raw = await self._client.chat.completions.with_raw_response.create( + model=model, + messages=[{"role": "user", "content": request.prompt}], + extra_body=extra_body or None, + extra_headers={ + # Ask OpenRouter to cache the successful response. A retry + # of the byte-identical request within the TTL is served + # from cache and billed at $0. + "X-OpenRouter-Cache": "true", + "X-OpenRouter-Cache-TTL": str(request.cache_ttl_seconds), + }, + ) + except APIStatusError as e: + raise_for_status( + e.status_code, _error_message(e.body) or e.message, e.response.headers + ) + # Connection errors and timeouts propagate as-is: Temporal retries them. + + payload = json.loads(raw.text) + error = payload.get("error") + if isinstance(error, dict): + # OpenRouter can return HTTP 200 with an error body and no choices + # when the upstream provider failed after the request was accepted. + raise_for_status( + int(error.get("code") or 500), _error_message(payload), raw.headers + ) + + choices = payload.get("choices") or [] + usage = payload.get("usage") or {} + cost = usage.get("cost") + result = OpenRouterResult( + prompt=request.prompt, + model=str(payload.get("model", model)), + answer=_content_to_text((choices[0].get("message") or {}).get("content")) + if choices + else "", + cost_usd=float(cost) if isinstance(cost, (int, float)) else 0.0, + generation_id=str(payload.get("id", "")), + cache_status=raw.headers.get("x-openrouter-cache-status", ""), + ) + activity.logger.info( + "OpenRouter call completed: attempt=%d model=%s cost_usd=%.6f cache=%s id=%s", + activity.info().attempt, + result.model, + result.cost_usd, + result.cache_status or "-", + result.generation_id, + ) + + if request.fail_once_after_call and activity.info().attempt == 1: + # Demo hook: the Worker "crashes" after the response arrived. The + # retry re-sends the identical request and gets a cache hit. + raise ApplicationError( + "Simulated failure after the response was received", + type="SimulatedFailure", + ) + + return result +``` + + +Three details in that Activity carry most of the value. + +### Classify errors + +OpenRouter's error codes tell you whether a retry can help. The Activity turns them into an `ApplicationError` with the +matching retry posture, and passes a `Retry-After` header through as the next retry delay: + +| Status | Meaning | Retry? | +|---|---|---| +| 408, 429 | Timeout, rate limited (`Retry-After` may be set) | Yes, honoring `Retry-After` | +| 500, 502, 503, 524, 529 | Server error, model down, no provider available, edge timeout, provider overloaded | Yes | +| 400 | Bad request | No | +| 401 | Invalid API key | No | +| 402 | Insufficient credits on the API key | No; see [Pause when money runs out](#pause-when-money-runs-out) | +| 403 | Moderation or permission block | No | + +OpenRouter can also return HTTP 200 with an `error` object in the body and no `choices` when the upstream provider +failed after the request was accepted. The Activity reads the raw body first and classifies that case by the code +inside the error. + +### Make retries free with response caching + +An Activity is at-least-once. If a Worker dies after OpenRouter answered but before Temporal recorded the result, +Temporal retries the Activity and sends the request again. The Activity sends OpenRouter's +`X-OpenRouter-Cache: true` header, so OpenRouter serves the retried, byte-identical request from its response cache +and bills $0 for it. OpenRouter writes the cache shortly after the response completes; a retry that arrives before +that write lands is a `MISS` and is billed. The response header `X-OpenRouter-Cache-Status` reports `HIT` or `MISS`, and the sample returns it +with each result. + +For this to work, nothing per-attempt may go in the request body. The attempt number belongs in heartbeat details and +logs, not in the prompt. Two identical requests in flight at the same time both miss the cache and both bill. + +### Heartbeat + +A 60-second model call with a 90-second `start_to_close_timeout` would take 90 seconds to fail over after a Worker +crash. The Activity heartbeats, and the Workflow sets `heartbeat_timeout=timedelta(seconds=10)`, so Temporal notices +the crash in seconds. + +## Fan out a prompt batch + +The `prompt_batch` sample runs one Activity per prompt under a semaphore, so a slow or failing prompt never blocks the +rest. The Workflow records a prompt whose Activity fails with a non-retryable error as skipped rather than failing the batch: + + +[openrouter/prompt_batch/workflow.py](https://github.com/temporalio/samples-python/blob/main/openrouter/prompt_batch/workflow.py) +```py +semaphore = asyncio.Semaphore(batch.max_concurrency) +outcomes = await asyncio.gather( + *(self._answer(prompt, batch, semaphore) for prompt in batch.prompts) +) +``` + + +Each result carries the concrete model OpenRouter's [Auto Router](https://openrouter.ai/docs/guides/routing/routers/auto-router) +chose, OpenRouter's reported `usage.cost`, the generation ID, and the cache status. + +Each Activity adds a few events to Event History, and every answer is part of the Workflow result payload. The sample +caps a batch at 100 prompts. For larger batches, start one Workflow per slice, or use the +[sliding window](https://github.com/temporalio/samples-python/tree/main/batch_sliding_window) pattern with +Continue-As-New. + +## Pause when money runs out + +The `budget_gate` sample is the same batch, except that it pauses instead of failing when money runs out, and resumes +when a person raises the budget. Two things can pause it: + +- A soft budget in the Workflow input, checked against the cost OpenRouter reports on every response. The Workflow + reserves an estimate per in-flight call and parks a prompt when `spent + reserved + estimate` would exceed the + budget. +- OpenRouter returning 402 because the API key hit its credit limit. The Activity marks 402 non-retryable; the Workflow + catches it and parks that prompt instead of skipping it. + +Either way the prompt waits on `workflow.wait_condition`, which costs nothing while it waits and survives Worker +restarts. A `raise_budget` Update wakes every parked prompt. A `spend_report` Query shows the spend so far, the +reservations, the per-prompt ledger, and the parked prompts with the reason for each: + + +[openrouter/budget_gate/workflow.py](https://github.com/temporalio/samples-python/blob/main/openrouter/budget_gate/workflow.py) +```py +@workflow.update +def raise_budget(self, new_budget_usd: float) -> SpendReport: + """Raise the soft budget and wake every parked prompt. + + Send the current budget unchanged to resume after topping up credits + in the OpenRouter dashboard. + """ + self._budget_usd = new_budget_usd + self._budget_version += 1 + return self.spend_report() + +@raise_budget.validator +def validate_raise_budget(self, new_budget_usd: float) -> None: + if new_budget_usd < self._budget_usd: + raise ValueError( + f"New budget ${new_budget_usd} is below the current budget " + f"${self._budget_usd}; the budget can only go up." + ) + +@workflow.query +def spend_report(self) -> SpendReport: + return SpendReport( + budget_usd=self._budget_usd, + spent_usd=round(self._spent_usd, 6), + reserved_usd=round(self._reserved_usd, 6), + completed=len(self._ledger), + paused=dict(self._paused), + paused_reason=next(iter(self._paused.values()), None), + ledger=list(self._ledger), + ) +``` + + + +[openrouter/budget_gate/workflow.py](https://github.com/temporalio/samples-python/blob/main/openrouter/budget_gate/workflow.py) +```py +async def _reserve(self, prompt: str, estimate: float, timeout: timedelta) -> bool: + """Reserve `estimate` against the budget, parking until it fits.""" + + def fits() -> bool: + return self._spent_usd + self._reserved_usd + estimate <= self._budget_usd + + if not fits(): + workflow.logger.info( + "Soft budget reached (spent $%.6f of $%.6f); pausing %r", + self._spent_usd, + self._budget_usd, + prompt, + ) + self._paused[prompt] = "soft_budget_exhausted" + try: + # Durable pause: survives Worker restarts and can wait for hours. + await workflow.wait_condition(fits, timeout=timeout) + except asyncio.TimeoutError: + return False + finally: + self._paused.pop(prompt, None) + self._reserved_usd += estimate + return True +``` + + +The validator rejects lowering the budget. Sending the current budget unchanged is how an operator says "I topped up +credits at OpenRouter"; it bumps a version counter that prompts parked on a 402 are waiting for. If nobody acts within +the approval timeout, the batch completes with the remaining prompts listed as skipped. + +The soft budget is a soft budget. The cost of a call is only known after the response. Overshoot is at most +`max_concurrency * estimated_cost_usd`, plus the gap between the estimate and the real cost of calls already in +flight. +To bound one call's cost, set `provider.max_price` in the request. The hard cap is the credit limit on the OpenRouter +API key, which is what produces the 402. + +Interacting with a paused batch from the Temporal CLI: + +```bash +temporal workflow query --workflow-id --type spend_report +temporal workflow update execute --workflow-id --name raise_budget --input '0.05' +``` + +## Routing: OpenRouter fallbacks and Temporal retries + +OpenRouter and Temporal both retry, at different time scales, for different reasons. Use both. + +| Concern | Where it lives | +|---|---| +| Provider outage or rate limit on one provider, within a request | OpenRouter: Auto Router, provider preferences, or a `models` list tried in order | +| Which model answers a given prompt | OpenRouter: `openrouter/auto` with a `cost_tier`, or an explicit model slug | +| A 429 with a 30-second `Retry-After` | Temporal: the Activity retries after 30 seconds, visibly | +| Worker crash mid-call | Temporal: Heartbeat Timeout, retry, cache hit | +| Waiting hours for a human to raise a budget or add credits | Temporal: `wait_condition` and an Update | +| Audit of every attempt, model, and cost | Temporal Event History plus OpenRouter's generation ids | + +Passing `models: ["a/first", "b/second"]` replaces the Auto Router: OpenRouter tries the list in order within one +request, and the response's `model` field reports which one answered. + +## Use OpenRouter with the OpenAI Agents SDK plugin + +OpenRouter speaks the OpenAI Chat Completions API, so the [OpenAI Agents SDK integration](openai-agents) can use it as +its model provider with no custom code. Point the stock `OpenAIProvider` at OpenRouter, turn off client retries, and +select Chat Completions: + + +[openai_agents/model_providers/run_openrouter_worker.py](https://github.com/temporalio/samples-python/blob/main/openai_agents/model_providers/run_openrouter_worker.py) +```py +def openrouter_provider() -> OpenAIProvider: + """OpenAI Agents SDK model provider backed by OpenRouter. + + OpenRouter speaks the OpenAI Chat Completions API, so the stock provider + works once it is pointed at OpenRouter's base URL. Client retries are off: + the plugin runs each model call as a Temporal Activity, and Temporal owns + the retries. + """ + default_headers: dict[str, str] = {} + # Optional app attribution for OpenRouter's rankings. + if referer := os.getenv("OPENROUTER_HTTP_REFERER"): + default_headers["HTTP-Referer"] = referer + if title := os.getenv("OPENROUTER_APP_TITLE"): + default_headers["X-OpenRouter-Title"] = title + + client = AsyncOpenAI( + base_url="https://openrouter.ai/api/v1", + api_key=os.environ["OPENROUTER_API_KEY"], + max_retries=0, + default_headers=default_headers or None, + ) + # Chat Completions is OpenRouter's primary endpoint; the Agents SDK + # defaults to the Responses API, which OpenRouter offers only in beta. + return OpenAIProvider(openai_client=client, use_responses=False) +``` + + +Pass the provider to `OpenAIAgentsPlugin(model_provider=...)` and set the agent's `model` to any OpenRouter model slug, +or to `openrouter/auto`. The plugin runs each model call as an Activity, so the Activity Retry Policy applies to +OpenRouter calls the same way it does to OpenAI calls. + +## Samples + +- [openrouter/prompt_batch](https://github.com/temporalio/samples-python/tree/main/openrouter/prompt_batch): fan out a + batch with the Auto Router, with a `--fail-once` flag that shows a retry served from cache at $0. +- [openrouter/budget_gate](https://github.com/temporalio/samples-python/tree/main/openrouter/budget_gate): the batch + that pauses on a soft budget or on 402 and resumes on `raise_budget`. +- [openai_agents/model_providers](https://github.com/temporalio/samples-python/tree/main/openai_agents/model_providers#openrouter): + OpenRouter as the model provider for an OpenAI Agents SDK agent. +- [samples-typescript/openrouter](https://github.com/temporalio/samples-typescript/tree/main/openrouter): the prompt + batch in TypeScript. diff --git a/sidebars.js b/sidebars.js index bcb90c2f2f..88a6951f14 100644 --- a/sidebars.js +++ b/sidebars.js @@ -728,6 +728,7 @@ const developPythonCategory = { 'develop/python/integrations/langgraph', 'develop/python/integrations/langsmith', 'develop/python/integrations/openai-agents', + 'develop/python/integrations/openrouter', 'develop/python/integrations/strands-agents', ], }, diff --git a/src/components/IntegrationsGrid/integrations-data.json b/src/components/IntegrationsGrid/integrations-data.json index d0c57735d1..aa9ef6ffdd 100644 --- a/src/components/IntegrationsGrid/integrations-data.json +++ b/src/components/IntegrationsGrid/integrations-data.json @@ -350,5 +350,23 @@ ], "sdk": "TypeScript", "href": "https://xmemory.ai/temporal/#quickstart--typescript" + }, + { + "name": "OpenRouter", + "description": "Call hundreds of models through OpenRouter from Temporal Activities, with durable retries, cost tracking, and budgets.", + "tags": [ + "Model gateway" + ], + "sdk": "Python", + "href": "/develop/python/integrations/openrouter" + }, + { + "name": "OpenRouter", + "description": "Call hundreds of models through OpenRouter from Temporal Activities, with durable retries and cost tracking.", + "tags": [ + "Model gateway" + ], + "sdk": "TypeScript", + "href": "https://github.com/temporalio/samples-typescript/tree/main/openrouter" } ] From 40220c6eaf30859574ad9be07da76ba062f484ef Mon Sep 17 00:00:00 2001 From: DABH Date: Mon, 14 Sep 2026 12:36:15 -0500 Subject: [PATCH 02/11] OpenRouter page: 403 key-limit is out of credits too; refresh snippets --- docs/develop/python/integrations/openrouter.mdx | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index 90cadf32e1..37ff88d7b9 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -179,8 +179,9 @@ matching retry posture, and passes a `Retry-After` header through as the next re | 500, 502, 503, 524, 529 | Server error, model down, no provider available, edge timeout, provider overloaded | Yes | | 400 | Bad request | No | | 401 | Invalid API key | No | -| 402 | Insufficient credits on the API key | No; see [Pause when money runs out](#pause-when-money-runs-out) | -| 403 | Moderation or permission block | No | +| 402 | The account is out of credits | No; raised as `OpenRouterOutOfCredits`, see [Pause when money runs out](#pause-when-money-runs-out) | +| 403 with `Key limit exceeded` | The API key hit its own credit limit | No; raised as `OpenRouterOutOfCredits` | +| 403, other | Moderation or permission block | No | OpenRouter can also return HTTP 200 with an `error` object in the body and no `choices` when the upstream provider failed after the request was accepted. The Activity reads the raw body first and classifies that case by the code @@ -235,8 +236,9 @@ when a person raises the budget. Two things can pause it: - A soft budget in the Workflow input, checked against the cost OpenRouter reports on every response. The Workflow reserves an estimate per in-flight call and parks a prompt when `spent + reserved + estimate` would exceed the budget. -- OpenRouter returning 402 because the API key hit its credit limit. The Activity marks 402 non-retryable; the Workflow - catches it and parks that prompt instead of skipping it. +- OpenRouter refusing the call for lack of credits: 402 when the account is out of credits, or 403 `Key limit exceeded` + when the API key hit its own limit. The Activity raises both as the non-retryable `OpenRouterOutOfCredits`; the + Workflow catches that type and parks the prompt instead of skipping it. Either way the prompt waits on `workflow.wait_condition`, which costs nothing while it waits and survives Worker restarts. A `raise_budget` Update wakes every parked prompt. A `spend_report` Query shows the spend so far, the @@ -315,7 +317,7 @@ The soft budget is a soft budget. The cost of a call is only known after the res `max_concurrency * estimated_cost_usd`, plus the gap between the estimate and the real cost of calls already in flight. To bound one call's cost, set `provider.max_price` in the request. The hard cap is the credit limit on the OpenRouter -API key, which is what produces the 402. +API key, which is what produces the `403 Key limit exceeded`. Interacting with a paused batch from the Temporal CLI: From ead1980866d5ef14adc2713ae96476b9bf22607f Mon Sep 17 00:00:00 2001 From: DABH Date: Wed, 30 Sep 2026 13:25:19 -0500 Subject: [PATCH 03/11] Integrations index: note sample-based integrations --- docs/develop/python/integrations/index.mdx | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/develop/python/integrations/index.mdx b/docs/develop/python/integrations/index.mdx index 9b9aca1835..e18048b8ba 100644 --- a/docs/develop/python/integrations/index.mdx +++ b/docs/develop/python/integrations/index.mdx @@ -13,6 +13,7 @@ import IntegrationsGrid from '@site/src/components/IntegrationsGrid'; The following integrations are available for the Temporal Python SDK. Most of these integrations are built on the Temporal Python SDK's [Plugin system](/develop/plugins-guide), which you -can also use to build your own integrations. +can also use to build your own integrations. A few, such as OpenRouter, are sample patterns that need nothing inside +Workflow code. From 0ef28d400d413eb5c0396a197134da38a1558222 Mon Sep 17 00:00:00 2001 From: DABH Date: Wed, 30 Sep 2026 13:44:14 -0500 Subject: [PATCH 04/11] OpenRouter page: refresh snippets after review changes --- .../python/integrations/openrouter.mdx | 40 +++++++++++++------ 1 file changed, 27 insertions(+), 13 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index 37ff88d7b9..578e7a74c7 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -283,7 +283,7 @@ def spend_report(self) -> SpendReport: [openrouter/budget_gate/workflow.py](https://github.com/temporalio/samples-python/blob/main/openrouter/budget_gate/workflow.py) ```py -async def _reserve(self, prompt: str, estimate: float, timeout: timedelta) -> bool: +async def _reserve(self, prompt: str, estimate: float) -> bool: """Reserve `estimate` against the budget, parking until it fits.""" def fits() -> bool: @@ -296,22 +296,35 @@ async def _reserve(self, prompt: str, estimate: float, timeout: timedelta) -> bo self._budget_usd, prompt, ) - self._paused[prompt] = "soft_budget_exhausted" - try: - # Durable pause: survives Worker restarts and can wait for hours. - await workflow.wait_condition(fits, timeout=timeout) - except asyncio.TimeoutError: + if not await self._park(prompt, "soft_budget_exhausted", fits): return False - finally: - self._paused.pop(prompt, None) self._reserved_usd += estimate return True + +async def _park(self, prompt: str, reason: str, until: Callable[[], bool]) -> bool: + """Durable pause until `until()` holds or the batch deadline passes. + + Survives Worker restarts and can wait for hours. Returns False when the + deadline passed first. + """ + remaining = self._deadline - workflow.now() + if remaining <= timedelta(0): + return False + self._paused[prompt] = reason + try: + await workflow.wait_condition(until, timeout=remaining) + return True + except asyncio.TimeoutError: + return False + finally: + self._paused.pop(prompt, None) ``` The validator rejects lowering the budget. Sending the current budget unchanged is how an operator says "I topped up -credits at OpenRouter"; it bumps a version counter that prompts parked on a 402 are waiting for. If nobody acts within -the approval timeout, the batch completes with the remaining prompts listed as skipped. +credits at OpenRouter"; it bumps a version counter that prompts parked on an out-of-credits error are waiting for. If +nobody acts before the batch's approval deadline (one deadline shared by every parked prompt), the batch completes +with the remaining prompts listed as skipped. The soft budget is a soft budget. The cost of a call is only known after the response. Overshoot is at most `max_concurrency * estimated_cost_usd`, plus the gap between the estimate and the real cost of calls already in @@ -378,9 +391,10 @@ def openrouter_provider() -> OpenAIProvider: ``` -Pass the provider to `OpenAIAgentsPlugin(model_provider=...)` and set the agent's `model` to any OpenRouter model slug, -or to `openrouter/auto`. The plugin runs each model call as an Activity, so the Activity Retry Policy applies to -OpenRouter calls the same way it does to OpenAI calls. +Pass the provider to `OpenAIAgentsPlugin(model_provider=...)` and set the agent's `model` to `openrouter/auto` or any +OpenRouter model slug. The plugin runs each model call as an Activity, so the Activity Retry Policy applies to +OpenRouter calls the same way it does to OpenAI calls. Tools that run inside the Workflow must be `async`: the Agents +SDK runs sync tools in a thread, which the Workflow sandbox does not allow. ## Samples From 3ec1805e8f9f656296be8471db9e7498f2e636a2 Mon Sep 17 00:00:00 2001 From: DABH Date: Thu, 1 Oct 2026 01:20:51 -0500 Subject: [PATCH 05/11] OpenRouter page: reported spend wording; refresh snippets --- .../python/integrations/openrouter.mdx | 20 ++++++++++++++----- 1 file changed, 15 insertions(+), 5 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index 578e7a74c7..f081045dd0 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -135,21 +135,25 @@ async def _send(self, request: OpenRouterRequest) -> OpenRouterResult: choices = payload.get("choices") or [] usage = payload.get("usage") or {} cost = usage.get("cost") + if not isinstance(cost, (int, float)): + # OpenRouter reports cost on every response; if it is ever missing, + # say so rather than pretending the call was free. + activity.logger.warning("OpenRouter response has no usage.cost") result = OpenRouterResult( prompt=request.prompt, model=str(payload.get("model", model)), answer=_content_to_text((choices[0].get("message") or {}).get("content")) if choices else "", - cost_usd=float(cost) if isinstance(cost, (int, float)) else 0.0, + cost_usd=float(cost) if isinstance(cost, (int, float)) else None, generation_id=str(payload.get("id", "")), cache_status=raw.headers.get("x-openrouter-cache-status", ""), ) activity.logger.info( - "OpenRouter call completed: attempt=%d model=%s cost_usd=%.6f cache=%s id=%s", + "OpenRouter call completed: attempt=%d model=%s cost_usd=%s cache=%s id=%s", activity.info().attempt, result.model, - result.cost_usd, + "unknown" if result.cost_usd is None else f"{result.cost_usd:.6f}", result.cache_status or "-", result.generation_id, ) @@ -221,7 +225,9 @@ outcomes = await asyncio.gather( Each result carries the concrete model OpenRouter's [Auto Router](https://openrouter.ai/docs/guides/routing/routers/auto-router) -chose, OpenRouter's reported `usage.cost`, the generation ID, and the cache status. +chose, OpenRouter's reported `usage.cost`, the generation ID, and the cache status. The batch's `reported_cost_usd` +sums what OpenRouter reported on each prompt's final, successful attempt. It is not a bill: an attempt that was billed +but whose response never reached Temporal is not in it. OpenRouter's dashboard is the source of truth for spend. Each Activity adds a few events to Event History, and every answer is part of the Workflow result payload. The sample caps a batch at 100 prompts. For larger batches, start one Workflow per slice, or use the @@ -260,6 +266,8 @@ def raise_budget(self, new_budget_usd: float) -> SpendReport: @raise_budget.validator def validate_raise_budget(self, new_budget_usd: float) -> None: + if not math.isfinite(new_budget_usd): + raise ValueError("The budget must be a finite number.") if new_budget_usd < self._budget_usd: raise ValueError( f"New budget ${new_budget_usd} is below the current budget " @@ -326,7 +334,9 @@ credits at OpenRouter"; it bumps a version counter that prompts parked on an out nobody acts before the batch's approval deadline (one deadline shared by every parked prompt), the batch completes with the remaining prompts listed as skipped. -The soft budget is a soft budget. The cost of a call is only known after the response. Overshoot is at most +The soft budget is a soft budget. It counts what OpenRouter reported on final attempts, charging the estimate when a +response carries no cost, so it tracks reported spend rather than the bill. The cost of a call is only known after the +response. Overshoot is at most `max_concurrency * estimated_cost_usd`, plus the gap between the estimate and the real cost of calls already in flight. To bound one call's cost, set `provider.max_price` in the request. The hard cap is the credit limit on the OpenRouter From 9a890ab95cb08c2e9ba019773f999f9756172f60 Mon Sep 17 00:00:00 2001 From: DABH Date: Thu, 1 Oct 2026 13:10:35 -0500 Subject: [PATCH 06/11] OpenRouter page: say what Event History records per Activity; refresh snippets --- docs/develop/python/integrations/openrouter.mdx | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index f081045dd0..cadbffcbdb 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -17,13 +17,13 @@ of models from many providers. It picks the provider and model for each request, what every response cost. Temporal handles everything around those calls. Each call runs as an Activity, so it gets retries with backoff, a -timeout, and a durable record in Event History of what was called and what it cost. The Workflow around the +timeout, and a durable record in Event History of its result, its cost, and how many attempts it took. The Workflow around the Activities can fan out over a batch with bounded concurrency, survive a Worker crash without re-running finished calls, and pause for hours until a person acts. The division of labor is simple. OpenRouter decides *which provider and model* serve a request, in milliseconds. Temporal decides *what happens over time*: waiting out a rate limit, surviving a crash, parking until someone raises a -budget, and keeping the audit trail. +budget, and keeping the record of what happened. This integration is a sample pattern rather than a plugin: OpenRouter needs nothing inside Workflow code. Code snippets in this guide are taken from the [OpenRouter samples](https://github.com/temporalio/samples-python/tree/main/openrouter). @@ -59,8 +59,9 @@ up to an hour, which hides attempts from Temporal. ## Call OpenRouter from an Activity Build one client for the Worker's lifetime with client-side retries turned off, and pass it to the Activity class. The -Activity makes exactly one HTTP call per attempt, so every attempt is visible in Event History and the Activity's Retry -Policy is the only retry policy in play. +Activity makes exactly one HTTP call per attempt, so the Activity's Retry Policy is the only retry policy in play. +Event History records each Activity's result, attempt count, and last failure; the sample logs each attempt's model, +cost, and cache status from the Worker. ```python from openai import AsyncOpenAI @@ -297,7 +298,10 @@ async def _reserve(self, prompt: str, estimate: float) -> bool: def fits() -> bool: return self._spent_usd + self._reserved_usd + estimate <= self._budget_usd - if not fits(): + # Loop rather than check once: when the budget is raised, every parked + # prompt is woken before any of them runs, so each must re-check after + # waking in case an earlier one already took the new headroom. + while not fits(): workflow.logger.info( "Soft budget reached (spent $%.6f of $%.6f); pausing %r", self._spent_usd, @@ -360,7 +364,7 @@ OpenRouter and Temporal both retry, at different time scales, for different reas | A 429 with a 30-second `Retry-After` | Temporal: the Activity retries after 30 seconds, visibly | | Worker crash mid-call | Temporal: Heartbeat Timeout, retry, cache hit | | Waiting hours for a human to raise a budget or add credits | Temporal: `wait_condition` and an Update | -| Audit of every attempt, model, and cost | Temporal Event History plus OpenRouter's generation ids | +| Record of each call's result, cost, and retries | Temporal Event History (result, attempt count, last failure), Worker logs per attempt, OpenRouter's generation IDs | Passing `models: ["a/first", "b/second"]` replaces the Auto Router: OpenRouter tries the list in order within one request, and the response's `model` field reports which one answered. From 1fd7bd3a272f4ceea3827ee6ecf32f162379f69a Mon Sep 17 00:00:00 2001 From: DABH Date: Thu, 1 Oct 2026 13:28:01 -0500 Subject: [PATCH 07/11] OpenRouter page: transient 402, choice-level errors, overshoot wording; refresh snippets --- .../python/integrations/openrouter.mdx | 51 ++++++++++--------- 1 file changed, 27 insertions(+), 24 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index cadbffcbdb..b7a502ce6d 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -119,21 +119,24 @@ async def _send(self, request: OpenRouterRequest) -> OpenRouterResult: }, ) except APIStatusError as e: - raise_for_status( - e.status_code, _error_message(e.body) or e.message, e.response.headers - ) + error = _error_object(e.body) + error.setdefault("message", e.message) + raise_for_status(e.status_code, error, e.response.headers) # Connection errors and timeouts propagate as-is: Temporal retries them. payload = json.loads(raw.text) - error = payload.get("error") - if isinstance(error, dict): + if isinstance(payload.get("error"), dict): # OpenRouter can return HTTP 200 with an error body and no choices # when the upstream provider failed after the request was accepted. - raise_for_status( - int(error.get("code") or 500), _error_message(payload), raw.headers - ) - + error = _error_object(payload) + raise_for_status(_error_code(error, 500), error, raw.headers) choices = payload.get("choices") or [] + choice_error = choices[0].get("error") if choices else None + if isinstance(choice_error, dict): + # Or a 200 with a partial answer and the provider's error on the + # choice itself; a partial answer is not an answer. + raise_for_status(_error_code(choice_error, 500), choice_error, raw.headers) + usage = payload.get("usage") or {} cost = usage.get("cost") if not isinstance(cost, (int, float)): @@ -184,13 +187,14 @@ matching retry posture, and passes a `Retry-After` header through as the next re | 500, 502, 503, 524, 529 | Server error, model down, no provider available, edge timeout, provider overloaded | Yes | | 400 | Bad request | No | | 401 | Invalid API key | No | -| 402 | The account is out of credits | No; raised as `OpenRouterOutOfCredits`, see [Pause when money runs out](#pause-when-money-runs-out) | -| 403 with `Key limit exceeded` | The API key hit its own credit limit | No; raised as `OpenRouterOutOfCredits` | +| 402 with `error.metadata.limit_source` of `openrouter_in_flight_budget` | In-flight budget cap, transient | Yes, honoring `Retry-After` | +| 402, other | The account or API key is out of credits | No; raised as `OpenRouterOutOfCredits`, see [Pause when money runs out](#pause-when-money-runs-out) | +| 403 with `Key limit exceeded` | A per-key limit, as observed in practice | No; raised as `OpenRouterOutOfCredits` | | 403, other | Moderation or permission block | No | OpenRouter can also return HTTP 200 with an `error` object in the body and no `choices` when the upstream provider -failed after the request was accepted. The Activity reads the raw body first and classifies that case by the code -inside the error. +failed after the request was accepted, or with a partial answer and an `error` on the choice. The Activity reads the +raw body first and classifies both cases by the code inside the error. ### Make retries free with response caching @@ -243,9 +247,9 @@ when a person raises the budget. Two things can pause it: - A soft budget in the Workflow input, checked against the cost OpenRouter reports on every response. The Workflow reserves an estimate per in-flight call and parks a prompt when `spent + reserved + estimate` would exceed the budget. -- OpenRouter refusing the call for lack of credits: 402 when the account is out of credits, or 403 `Key limit exceeded` - when the API key hit its own limit. The Activity raises both as the non-retryable `OpenRouterOutOfCredits`; the - Workflow catches that type and parks the prompt instead of skipping it. +- OpenRouter refusing the call for lack of credits: a 402 for the account or the API key, or the 403 `Key limit + exceeded` seen in practice for a per-key limit. The Activity raises both as the non-retryable + `OpenRouterOutOfCredits`; the Workflow catches that type and parks the prompt instead of skipping it. Either way the prompt waits on `workflow.wait_condition`, which costs nothing while it waits and survives Worker restarts. A `raise_budget` Update wakes every parked prompt. A `spend_report` Query shows the spend so far, the @@ -283,7 +287,6 @@ def spend_report(self) -> SpendReport: reserved_usd=round(self._reserved_usd, 6), completed=len(self._ledger), paused=dict(self._paused), - paused_reason=next(iter(self._paused.values()), None), ledger=list(self._ledger), ) ``` @@ -300,7 +303,8 @@ async def _reserve(self, prompt: str, estimate: float) -> bool: # Loop rather than check once: when the budget is raised, every parked # prompt is woken before any of them runs, so each must re-check after - # waking in case an earlier one already took the new headroom. + # waking in case an earlier one already took the new headroom. (Each + # re-park starts a new timer for the remaining time; fine at this scale.) while not fits(): workflow.logger.info( "Soft budget reached (spent $%.6f of $%.6f); pausing %r", @@ -340,11 +344,10 @@ with the remaining prompts listed as skipped. The soft budget is a soft budget. It counts what OpenRouter reported on final attempts, charging the estimate when a response carries no cost, so it tracks reported spend rather than the bill. The cost of a call is only known after the -response. Overshoot is at most -`max_concurrency * estimated_cost_usd`, plus the gap between the estimate and the real cost of calls already in -flight. +response, and reservations count against the budget, so overshoot is at most `max_concurrency` times how far a +call's real cost exceeds the estimate. To bound one call's cost, set `provider.max_price` in the request. The hard cap is the credit limit on the OpenRouter -API key, which is what produces the `403 Key limit exceeded`. +API key, which is what produces the out-of-credits error. Interacting with a paused batch from the Temporal CLI: @@ -399,8 +402,8 @@ def openrouter_provider() -> OpenAIProvider: max_retries=0, default_headers=default_headers or None, ) - # Chat Completions is OpenRouter's primary endpoint; the Agents SDK - # defaults to the Responses API, which OpenRouter offers only in beta. + # These samples use Chat Completions, OpenRouter's primary endpoint; the + # Agents SDK defaults to the Responses API, which OpenRouter also offers. return OpenAIProvider(openai_client=client, use_responses=False) ``` From 1d0d9e737dfa58165d68d710081e0ede59629ae9 Mon Sep 17 00:00:00 2001 From: DABH Date: Thu, 1 Oct 2026 13:39:23 -0500 Subject: [PATCH 08/11] OpenRouter page: refresh snippets; note the Retry-After cap --- docs/develop/python/integrations/openrouter.mdx | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index b7a502ce6d..c7b426dc8b 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -136,6 +136,9 @@ async def _send(self, request: OpenRouterRequest) -> OpenRouterResult: # Or a 200 with a partial answer and the provider's error on the # choice itself; a partial answer is not an answer. raise_for_status(_error_code(choice_error, 500), choice_error, raw.headers) + if not choices: + # No error and no answer: treat like a server error and retry. + raise_for_status(500, {"message": "Response has no choices"}, raw.headers) usage = payload.get("usage") or {} cost = usage.get("cost") @@ -183,7 +186,7 @@ matching retry posture, and passes a `Retry-After` header through as the next re | Status | Meaning | Retry? | |---|---|---| -| 408, 429 | Timeout, rate limited (`Retry-After` may be set) | Yes, honoring `Retry-After` | +| 408, 429 | Timeout, rate limited (`Retry-After` may be set) | Yes, honoring `Retry-After` up to five minutes | | 500, 502, 503, 524, 529 | Server error, model down, no provider available, edge timeout, provider overloaded | Yes | | 400 | Bad request | No | | 401 | Invalid API key | No | From 1d04b6b5ed9e63e3e39b50c28d6c6c29c7e6ac0c Mon Sep 17 00:00:00 2001 From: DABH Date: Sun, 4 Oct 2026 14:55:16 -0500 Subject: [PATCH 09/11] OpenRouter page: refresh snippets --- docs/develop/python/integrations/openrouter.mdx | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index c7b426dc8b..cf642ca84c 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -302,7 +302,11 @@ async def _reserve(self, prompt: str, estimate: float) -> bool: """Reserve `estimate` against the budget, parking until it fits.""" def fits() -> bool: - return self._spent_usd + self._reserved_usd + estimate <= self._budget_usd + # Tolerance absorbs float accumulation; see BUDGET_TOLERANCE_USD. + return ( + self._spent_usd + self._reserved_usd + estimate + <= self._budget_usd + BUDGET_TOLERANCE_USD + ) # Loop rather than check once: when the budget is raised, every parked # prompt is woken before any of them runs, so each must re-check after From aee1b4526ae94786cdb18de7c855583e2ea00245 Mon Sep 17 00:00:00 2001 From: DABH Date: Sun, 4 Oct 2026 15:02:59 -0500 Subject: [PATCH 10/11] OpenRouter page: mention unknown_cost_count --- docs/develop/python/integrations/openrouter.mdx | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index cf642ca84c..1ae63dae07 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -234,8 +234,9 @@ outcomes = await asyncio.gather( Each result carries the concrete model OpenRouter's [Auto Router](https://openrouter.ai/docs/guides/routing/routers/auto-router) chose, OpenRouter's reported `usage.cost`, the generation ID, and the cache status. The batch's `reported_cost_usd` -sums what OpenRouter reported on each prompt's final, successful attempt. It is not a bill: an attempt that was billed -but whose response never reached Temporal is not in it. OpenRouter's dashboard is the source of truth for spend. +sums what OpenRouter reported on each prompt's final, successful attempt, and `unknown_cost_count` is how many results +had no cost. It is not a bill: an attempt that was billed but whose response never reached Temporal is not in it. +OpenRouter's dashboard is the source of truth for spend. Each Activity adds a few events to Event History, and every answer is part of the Workflow result payload. The sample caps a batch at 100 prompts. For larger batches, start one Workflow per slice, or use the From c6e570c5260bae8722c9406525f5383e12dce8bb Mon Sep 17 00:00:00 2001 From: DABH Date: Sun, 4 Oct 2026 15:03:44 -0500 Subject: [PATCH 11/11] OpenRouter page: refresh snippets --- docs/develop/python/integrations/openrouter.mdx | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/develop/python/integrations/openrouter.mdx b/docs/develop/python/integrations/openrouter.mdx index 1ae63dae07..8ca5c99d0a 100644 --- a/docs/develop/python/integrations/openrouter.mdx +++ b/docs/develop/python/integrations/openrouter.mdx @@ -331,6 +331,9 @@ async def _park(self, prompt: str, reason: str, until: Callable[[], bool]) -> bo Survives Worker restarts and can wait for hours. Returns False when the deadline passed first. """ + if until(): + # Nothing to wait for (a top-up already landed during the call). + return True remaining = self._deadline - workflow.now() if remaining <= timedelta(0): return False