Repository navigation
Expand file tree
/
Copy pathshared.py
More file actions
145 lines (120 loc) · 5.1 KB
/
Copy pathshared.py
File metadata and controls
145 lines (120 loc) · 5.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
from dataclasses import dataclass, field
from datetime import timedelta
from typing import Optional
from temporalio.common import RetryPolicy
OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"
# OpenRouter's Auto Router picks a concrete model per request. The response's
# `model` field reports which one it chose.
DEFAULT_MODEL = "openrouter/auto"
PROMPT_BATCH_TASK_QUEUE = "openrouter-prompt-batch"
BUDGET_GATE_TASK_QUEUE = "openrouter-budget-gate"
# Each Activity adds a few events to the Workflow's Event History and each
# answer is stored in the Workflow result payload. Keep batches small enough to
# stay well under the history and payload limits; see the README for the
# sliding-window pattern for larger batches.
MAX_PROMPTS_PER_BATCH = 100
# A parked batch can wait at most this long (30 days) for a raise_budget.
MAX_APPROVAL_TIMEOUT_SECONDS = 30 * 24 * 3600
# Budget comparisons are done on floats that accumulate per-call costs, so
# allow a hair of slack: ten calls at $0.001 must fit a $0.01 budget.
BUDGET_TOLERANCE_USD = 1e-9
# Temporal owns retries: 1s, 2s, 4s, ... capped at 60s, five attempts. The
# Activity marks 4xx errors non-retryable and passes OpenRouter's Retry-After
# through as the next retry delay, so this policy only governs the rest.
OPENROUTER_RETRY_POLICY = RetryPolicy(
initial_interval=timedelta(seconds=1),
backoff_coefficient=2.0,
maximum_interval=timedelta(seconds=60),
maximum_attempts=5,
)
@dataclass
class OpenRouterRequest:
"""One chat completion request. Everything here ends up in the request
body, so keep it free of per-attempt values (attempt number, timestamps):
OpenRouter's response cache keys on the exact body, and a retried attempt
should be byte-identical to the first one."""
prompt: str
model: str = DEFAULT_MODEL
# When set, sent as OpenRouter's `models` list and tried in order. This
# replaces the Auto Router.
fallback_models: list[str] = field(default_factory=list)
# Auto Router cost tier: low, medium, high, xhigh, or max. Only used with
# `openrouter/auto`.
cost_tier: str = "low"
# How long OpenRouter keeps a successful response cached so that a retry of
# the identical request is served for free.
cache_ttl_seconds: int = 600
# Demo hook: fail the first attempt *after* the response arrives, so the
# retry shows a cache hit billed at $0 in Event History.
fail_once_after_call: bool = False
@dataclass
class OpenRouterResult:
prompt: str
model: str
answer: str
# What OpenRouter reported for this attempt's response. None if the
# response carried no usage.cost (it always should).
cost_usd: Optional[float]
generation_id: str
# "HIT" or "MISS" from OpenRouter's X-OpenRouter-Cache-Status header, or ""
# when the header is absent.
cache_status: str
@dataclass
class SkippedPrompt:
prompt: str
reason: str
@dataclass
class BatchInput:
prompts: list[str]
model: str = DEFAULT_MODEL
max_concurrency: int = 5
fail_once_after_call: bool = False
@dataclass
class BatchResult:
results: list[OpenRouterResult]
skipped: list[SkippedPrompt]
# Sum of the cost OpenRouter reported on each prompt's final, successful
# attempt (budget_gate charges its estimate for a response with no cost).
# Attempts that were billed but whose result never reached Temporal (a
# Worker crash after the response, say) are not in here; OpenRouter's
# dashboard or /api/v1/key is the source of truth for spend.
reported_cost_usd: float
# How many successful prompts came back without a cost. When this is not
# zero, reported_cost_usd is a subtotal of the known costs (prompt_batch)
# or includes the estimate for those prompts (budget_gate).
unknown_cost_count: int = 0
@dataclass
class BudgetGateInput:
# The batch to run: prompts, model, concurrency. Same shape as prompt_batch.
batch: BatchInput
# Soft budget enforced by the Workflow from OpenRouter's reported cost.
budget_usd: float
# Reserved per in-flight call before its real cost is known. Reservations
# count against the budget, so overshoot is bounded by
# batch.max_concurrency * max(actual cost - estimate, 0): nothing if the
# estimate is high enough, unbounded if it is far too low.
estimated_cost_usd: float = 0.001
# How long, from the start of the batch, parked prompts wait for a
# `raise_budget` Update before the batch gives up on them. One deadline is
# shared by the whole batch.
approval_timeout_seconds: int = 3600
@dataclass
class LedgerEntry:
prompt: str
model: str
# Reported cost, or the batch's estimate when the response had no cost.
cost_usd: float
cost_known: bool
generation_id: str
cache_status: str
@dataclass
class SpendReport:
budget_usd: float
spent_usd: float
reserved_usd: float
completed: int
# Prompts currently parked, keyed by prompt text (duplicate prompts share
# one entry), with why: "soft_budget_exhausted" or "insufficient_credits"
# (OpenRouter refused the call for lack of credits).
paused: dict[str, str]
ledger: list[LedgerEntry]