From f38731f3649ca364c1eba0f6b0496d912fb8b125 Mon Sep 17 00:00:00 2001 From: Artur Shiriev Date: Thu, 1 Oct 2026 21:37:57 +0300 Subject: [PATCH] docs: fix stale facts and plain up user-facing prose Bring README, docs site, demos, SECURITY and contributing guide up to date with the circuit breaker and timeout middleware, fix misleading statements, and remove dash-heavy, bold-labelled, all-caps phrasing. --- README.md | 53 ++++--- SECURITY.md | 18 +-- docs/decoders.md | 43 +++--- docs/demos/bulkhead.md | 16 +- docs/demos/circuit-breaker.md | 27 ++-- docs/demos/engine.js | 8 +- docs/demos/full-stack.md | 29 ++-- docs/demos/index.md | 18 +-- docs/demos/retry.md | 33 ++-- docs/demos/timeout.md | 18 +-- docs/dev/contributing.md | 21 ++- docs/errors.md | 98 ++++++------ docs/index.md | 117 +++++++------- docs/middleware.md | 63 ++++---- docs/observability.md | 25 ++- docs/recipes/link-header-pagination.md | 20 +-- docs/recipes/modern-di.md | 51 +++---- docs/recipes/phase-decorator-patterns.md | 47 +++--- docs/resilience.md | 160 ++++++++++---------- docs/testing.md | 25 ++- mkdocs.yml | 4 +- pyproject.toml | 2 +- src/httpware/middleware/resilience/retry.py | 4 +- 23 files changed, 429 insertions(+), 471 deletions(-) diff --git a/README.md b/README.md index 79e118e..970ac6c 100644 --- a/README.md +++ b/README.md @@ -18,32 +18,33 @@ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff) [![ty](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ty/main/assets/badge/v0.json)](https://github.com/astral-sh/ty) -**Typed, resilient HTTP clients for Python — typed errors, typed response bodies, and composable resilience (retry, bulkhead, circuit breaker), sync or async.** +Typed, resilient HTTP clients for Python, sync and async. ## Why httpware -- **Errors you can catch by name** — a 404 raises `NotFoundError`, a 429 - `RateLimitedError`, automatically; everything else bubbles up under one - `httpware.StatusError` base. No `raise_for_status()`, no status-code - branching. -- **Typed response bodies** — `response_model=User` decodes the body straight - to your pydantic or msgspec type; a missing decoder fails fast, *before* the - request goes out. -- **Composable resilience** — retry + retry-budget, bulkhead, circuit breaker, - and timeout as middleware over standard `httpx2`. +- A 4xx or 5xx response raises an exception named after its status, such as + `NotFoundError` for 404 or `RateLimitedError` for 429. All of them subclass + `httpware.StatusError`, so you never call `raise_for_status()`. +- `response_model=User` decodes the body into your pydantic or msgspec type. + If no installed decoder handles the type, the call fails before the request + is sent. +- Retry with a retry budget, bulkhead, circuit breaker, and timeout ship as + middleware you compose per client. -Built on `httpx2`: httpware re-exports `httpx2.Request`/`httpx2.Response` and stays a thin wrapper, not a new HTTP abstraction. +httpware is a thin layer over `httpx2`: requests and responses are plain +`httpx2.Request` and `httpx2.Response` objects. > **Status:** Pre-1.0. Public API is subject to change between minor releases until v1.0. ## Install ```bash -pip install httpware # core only — no decoder -pip install httpware[pydantic] # + PydanticDecoder — BaseModel, dataclasses, primitives, generics -pip install httpware[msgspec] # + MsgspecDecoder — Struct, dataclasses, primitives, generics -pip install httpware[pydantic,msgspec] # both — BaseModel routes to pydantic, Struct to msgspec -pip install httpware[all] # everything (pydantic, msgspec, otel) +pip install httpware # core only, no decoder +pip install httpware[pydantic] # PydanticDecoder: BaseModel, dataclasses, primitives, generics +pip install httpware[msgspec] # MsgspecDecoder: Struct, dataclasses, primitives, generics +pip install httpware[pydantic,msgspec] # both; BaseModel goes to pydantic, Struct to msgspec +pip install httpware[otel] # OpenTelemetry span events +pip install httpware[all] # pydantic, msgspec, and otel ``` ## Quickstart @@ -71,22 +72,24 @@ async def main() -> None: asyncio.run(main()) ``` -The sync `Client` is identical — swap `AsyncClient` → `Client` and drop the `await` / `async with`. A 4xx/5xx response raises a typed `StatusError`; a malformed body raises `DecodeError`. Both subclass `httpware.ClientError`. +The sync `Client` works the same way: use `Client` instead of `AsyncClient`, and drop `await` and `async with`. A 4xx/5xx response raises a typed `StatusError`; a malformed body raises `DecodeError`. Both subclass `httpware.ClientError`. ## Documentation -Full guides live at **[httpware.modern-python.org](https://httpware.modern-python.org)**: +Full guides live at [httpware.modern-python.org](https://httpware.modern-python.org): -- **[Quickstart & observability](https://httpware.modern-python.org/)** — resilience middleware, streaming, and the stable logger/event contract. -- **[Middleware](https://httpware.modern-python.org/middleware/)** — write your own (auth, tracing, request-ID propagation). -- **[Resilience](https://httpware.modern-python.org/resilience/)** — retry + retry-budget, bulkhead, circuit breaker, timeout. -- **[Errors](https://httpware.modern-python.org/errors/)** — the exception tree and catching strategies. -- **[Testing](https://httpware.modern-python.org/testing/)** — `httpx2.MockTransport` injection. -- **[Recipes](https://httpware.modern-python.org/recipes/modern-di/)** — DI wiring, phase-decorator patterns, link-header pagination. +- [Quickstart](https://httpware.modern-python.org/): first requests, client options, streaming. +- [Resilience](https://httpware.modern-python.org/resilience/): retry and retry budget, bulkhead, circuit breaker, timeout. +- [Errors](https://httpware.modern-python.org/errors/): the exception tree and how to catch it. +- [Decoders](https://httpware.modern-python.org/decoders/): typed response bodies and custom decoders. +- [Middleware](https://httpware.modern-python.org/middleware/): writing your own (auth, tracing, request IDs). +- [Observability](https://httpware.modern-python.org/observability/): logger and event names, OpenTelemetry wiring. +- [Testing](https://httpware.modern-python.org/testing/): injecting `httpx2.MockTransport`. +- [Recipes](https://httpware.modern-python.org/recipes/modern-di/): DI wiring, phase decorators, Link header pagination. ## 🗒️ [Release notes](https://github.com/modern-python/httpware/releases) · 📦 [PyPI](https://pypi.org/project/httpware) · 📝 [License](https://github.com/modern-python/httpware/blob/main/LICENSE) ## Part of `modern-python` Browse the full list of templates and libraries in -[`modern-python`](https://github.com/modern-python) — see the org profile for the categorized index. +[`modern-python`](https://github.com/modern-python); the org profile has the categorized index. diff --git a/SECURITY.md b/SECURITY.md index 406ad16..8e8e5c6 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,18 +1,18 @@ -# Security Policy +# Security policy -## Reporting a Vulnerability +## Reporting a vulnerability If you discover a security vulnerability in `httpware`, please report it privately via [GitHub Security Advisories](https://github.com/modern-python/httpware/security/advisories/new). -**Do not file a public GitHub issue for security reports.** +Do not file a public GitHub issue for security reports. -## Disclosure Timeline +## Disclosure timeline -- We commit to acknowledging your report within **7 days**. -- We aim to provide a fix or detailed mitigation plan within **30 days** of confirmation. -- We follow a **90-day private disclosure window** before public disclosure of the vulnerability and fix, unless a coordinated earlier disclosure is in the interest of users (e.g., the vulnerability is already being actively exploited). +- We acknowledge reports within 7 days. +- We aim to provide a fix or a detailed mitigation plan within 30 days of confirming the report. +- We keep a vulnerability and its fix private for 90 days before disclosing them, unless an earlier coordinated disclosure serves users better, for example when the vulnerability is already being exploited. -## Supported Versions +## Supported versions Security fixes are provided for: @@ -31,5 +31,5 @@ In scope: Out of scope: -- Vulnerabilities in transitive dependencies (`httpx2`, `pydantic`, etc.) — report those upstream. We will fast-track a `httpware` release pinning the patched version once an upstream fix is available. +- Vulnerabilities in dependencies such as `httpx2` or `pydantic`. Report those upstream; once a fix is released there, we will quickly publish an httpware release that requires the patched version. - Misconfiguration in consuming applications. diff --git a/docs/decoders.md b/docs/decoders.md index 098edfd..5b094fc 100644 --- a/docs/decoders.md +++ b/docs/decoders.md @@ -1,8 +1,8 @@ # Decoders -`httpware`'s typed-response extension point is the **`ResponseDecoder` protocol**. A decoder turns raw response bytes into a typed object: when you pass `response_model=` to `send` / `send_with_response`, the client walks its decoder list, picks the first one that claims your model, and hands it the body. +A decoder turns raw response bytes into a typed object. When you pass `response_model=` to a request, the client goes through its decoder list, picks the first decoder that accepts your type, and hands it the body. -The built-in `PydanticDecoder` and `MsgspecDecoder` are themselves implementations of this protocol; nothing about them is privileged. Reach for a custom decoder when you need a body **format** the built-ins don't speak (CSV, XML, MessagePack, a bespoke binary frame) or a **type system** they don't cover (`attrs`, `marshmallow`, your own class hierarchy). If pydantic or msgspec already decodes your model, you don't need one — see [When NOT to write a decoder](#when-not-to-write-a-decoder). +`PydanticDecoder` and `MsgspecDecoder` implement the same `ResponseDecoder` protocol you would. Write your own when the body is in a format the built-ins can't read (CSV, XML, MessagePack, a custom binary format) or the type comes from a library they don't support (`attrs`, `marshmallow`, your own classes). If pydantic or msgspec already decodes your type, you don't need one; see [When not to write a decoder](#when-not-to-write-a-decoder). ## The protocol @@ -20,30 +20,29 @@ class ResponseDecoder(Protocol): def decode(self, content: bytes, model: type[T]) -> T: ... ``` -Two methods, two distinct jobs: +`can_decode(model)` decides whether the decoder takes a type. The client asks each decoder in `decoders=[...]` order and uses the first one that returns `True`. Accept every type you can handle and let the list order express the caller's preference, but reject types that belong to another library: a CSV decoder should not accept a `pydantic.BaseModel`. `can_decode` must never raise. It runs before the request and outside the `DecodeError` wrapping that protects `decode`, so an exception there reaches the caller as something other than a `ClientError`. If the decoder can't tell, return `False`. -- **`can_decode(model) -> bool`** — the dispatch predicate. The client walks `decoders=[...]` in order and picks the **first** decoder that returns `True`. Claim every model you can actually handle (broad is correct — list ordering, not narrow predicates, encodes the caller's preference), but **reject another library's native types**: a CSV decoder has no business claiming a `pydantic.BaseModel`. `can_decode` **MUST NOT raise** — it runs at dispatch time, before the HTTP call and *outside* the `DecodeError` wrap that protects `decode`, so an exception here escapes `httpware`'s `ClientError` contract instead of being translated. A decoder that can't decide must return `False` (decline), not raise. -- **`decode(content, model) -> T`** — the decode itself, raw response bytes in, a `model` instance out. Any exception you raise here is caught by the client and wrapped as `httpware.DecodeError` (carrying `response`, `model`, and the `original` exception). You do **not** need to raise `DecodeError` yourself — raise whatever your parser raises and let the seam translate it. +`decode(content, model)` takes the raw body and returns an instance of `model`. The client wraps any exception it raises in `httpware.DecodeError`, with `response`, `model` and the `original` exception attached, so raise whatever your parser raises. -The protocol is `@runtime_checkable` and structural: any object with these two methods satisfies it. You do not subclass anything. +The protocol is structural and `@runtime_checkable`: any object with these two methods satisfies it, with no base class. ## How the client resolves a model -Both clients take `decoders: Sequence[ResponseDecoder] | None = None`, composed once at `__init__` and frozen for the client's lifetime. +Both clients take `decoders: Sequence[ResponseDecoder] | None = None`. The list is fixed when the client is built. -- **Order is preference.** `decoders=[CsvDecoder(), PydanticDecoder()]` asks the CSV decoder first; pydantic only sees models CSV declined. List position is how you disambiguate a shape two decoders could both claim. -- **`decoders=None`** resolves against installed extras — pydantic-first when both are present, either-only when one is, an empty tuple when neither. To *add* a decoder without losing the built-ins, list them explicitly: `decoders=[CsvDecoder(), PydanticDecoder()]`. -- **No claimer is a pre-flight error.** When `response_model=` is set and no decoder claims it, the client raises `MissingDecoderError` **before** sending the request — you find out at wiring time, not after a wasted round-trip. This is distinct from `DecodeError`: `MissingDecoderError` means *nothing handles this model* (fix: install an extra or pass `decoders=[...]`); `DecodeError` means *a decoder ran and the payload was malformed* (fix: the server or the model). See [Errors](errors.md). +- Order is preference. With `decoders=[CsvDecoder(), PydanticDecoder()]`, pydantic only sees the types the CSV decoder declined. When two decoders could both handle a type, the earlier one wins. +- `decoders=None` uses the installed extras: pydantic then msgspec when both are installed, whichever one is installed, or no decoders at all. Passing a list replaces the defaults, so include the built-ins you still want: `decoders=[CsvDecoder(), PydanticDecoder()]`. +- If `response_model=` is set and no decoder accepts it, the client raises `MissingDecoderError` before sending the request. That means nothing handles the type, and the fix is to install an extra or pass `decoders=[...]`. A `DecodeError` instead means a decoder ran and the body didn't fit the type, which points at the server or the model. See [Errors](errors.md). -## Decoders are sync — for both clients +## One sync protocol for both clients -Unlike middleware, which has separate `AsyncMiddleware` and `Middleware` flavors, there is **one** `ResponseDecoder` protocol, shared by `AsyncClient` and `Client` alike. `decode` is a synchronous method: by the time it runs, the body has already been read off the wire, so decoding is pure CPU work with nothing to await. Write one decoder and pass it to either client. +Middleware comes in sync and async versions, but there is only one `ResponseDecoder` protocol, used by both clients. `decode` is synchronous because the body has already been read when it runs, so there is nothing to await. The same decoder works with either client. ## Writing your own ### Worked example: a CSV decoder -A decoder for `text/csv` endpoints that returns a `list` of dataclass rows. Both built-ins are JSON, so this is the case they can't cover — and it shows the seam's real shape: raw bytes in, typed object out, no JSON anywhere. +This decoder reads `text/csv` responses into a `list` of dataclass rows. Both built-ins only read JSON, so they can't handle this case. ```python import csv @@ -77,7 +76,7 @@ class CsvDecoder: return [row_type(**{name: field_types[name](value) for name, value in row.items()}) for row in reader] ``` -`can_decode` is total and never raises: a non-`list` model, a bare `list`, or `list[int]` all fall through to `False`. `decode` coerces each CSV cell with its field's type (CSV values arrive as strings) — a real decoder would handle optionals, dates, and missing columns; this is where your domain logic goes. Wire it ahead of the built-ins so it gets first refusal on `list[...]` models while pydantic still handles everything else: +`can_decode` never raises: a non-`list` type, a bare `list` and `list[int]` all return `False`. `decode` converts each CSV cell, which arrives as a string, with its field's type. A real decoder would also handle optional fields, dates and missing columns. Put it before the built-ins so it sees `list[...]` types first, while pydantic still handles everything else: ```python @dataclasses.dataclass @@ -103,7 +102,7 @@ The same decoder instance works with a sync `Client(decoders=[CsvDecoder(), Pyda ### A note on claiming the right models -`can_decode` is a contract with the *rest of the list*. Claim too broadly and you steal models from decoders behind you; claim too narrowly and your decoder never runs. The rule of thumb: claim exactly the types you natively own, and reject another library's. An adapter for a third-party type system narrows its claim to that system — for example, a [`cattrs`](https://catt.rs)-backed decoder for `attrs` classes: +`can_decode` affects the other decoders in the list. Accept too much and you take types from the decoders after yours; accept too little and yours never runs. Accept exactly the types your decoder is for, and reject types from other libraries. A decoder for a third-party type system should accept only that system's types, as in this [`cattrs`](https://catt.rs) decoder for `attrs` classes: ```python import json @@ -122,15 +121,15 @@ class CattrsDecoder: return self._converter.structure(json.loads(content), model) ``` -Note this decoder is **two-pass** (`json.loads`, then `structure`). The built-in adapters deliberately decode in a single bytes-in pass (`TypeAdapter.validate_json`, `msgspec.json.Decoder.decode`) to skip the intermediate `dict` allocation — but that's a *performance choice for the built-ins*, not a protocol obligation. A custom decoder may go two-pass when its underlying library only structures from native Python objects; you pay one extra allocation, nothing more. +This decoder makes two passes: `json.loads`, then `structure`. The built-ins decode straight from bytes (`TypeAdapter.validate_json`, `msgspec.json.Decoder.decode`) to avoid building an intermediate `dict`, but that is only an optimization. Two passes are fine when your library can only work from Python objects; the cost is one extra allocation. -### When NOT to write a decoder +### When not to write a decoder -- **Your model is JSON.** Dataclasses, `TypedDict`s, primitives, pydantic models, and msgspec `Struct`s are all covered by the built-in `PydanticDecoder` / `MsgspecDecoder`. Install the extra (`httpware[pydantic]` or `httpware[msgspec]`) instead of writing a decoder. -- **You only want raw bytes or text.** Don't pass `response_model=` at all — call `send` (or a verb method) without it and read `response.content` / `response.text` directly. Decoders are for *typed* bodies. -- **The transform is per-call, not per-type.** If the shaping depends on the request rather than the model, it's a [middleware](middleware.md) concern, not a decoder. +- The body is JSON. `PydanticDecoder` and `MsgspecDecoder` handle dataclasses, `TypedDict`s, primitives, pydantic models and msgspec `Struct`s. Install `httpware[pydantic]` or `httpware[msgspec]`. +- You want raw bytes or text. Leave out `response_model=` and read `response.content` or `response.text`. +- The shaping depends on the request rather than the type. That belongs in [middleware](middleware.md). ## See also -- **`src/httpware/decoders/pydantic.py` and `msgspec.py`** — the built-in adapters as reference implementations, including how they memoize a `can_decode` verdict and cache the underlying parser per model. -- **[Quick-Start: typed responses](index.md)** — composing `response_model=` with the default decoder list. +- [`src/httpware/decoders/`](https://github.com/modern-python/httpware/tree/main/src/httpware/decoders): the built-in decoders, including how they cache `can_decode` results and parsers per type. +- [Quickstart: typed responses](index.md#typed-responses): `response_model=` with the default decoders. diff --git a/docs/demos/bulkhead.md b/docs/demos/bulkhead.md index 18e0fe1..e2ebf66 100644 --- a/docs/demos/bulkhead.md +++ b/docs/demos/bulkhead.md @@ -1,8 +1,8 @@ # Bulkhead -One slow dependency can sink a whole client: if every worker blocks on the slow call, -fast calls starve behind them. A bulkhead caps concurrency to that dependency — excess -calls fail fast with `BulkheadFullError` instead of piling up and exhausting the client. +One slow dependency can stall a whole client: workers block on the slow calls and fast +calls wait behind them. A bulkhead caps how many calls to that dependency run at once, +and calls over the cap fail fast with `BulkheadFullError` instead of queueing.
@@ -17,13 +17,13 @@ document.addEventListener('DOMContentLoaded', function () { ], buildStops: () => [ { when: (s) => s.now >= 1.2, spot: ['ifA', 'poolB'], title: 'Fast calls, healthy pool', - body: 'Both clients are humming. The httpware client has a bulkhead: at most 8 calls to this dependency at once.' }, + body: 'Both clients are healthy. The httpware client has a bulkhead that allows at most 8 calls to this dependency at once.' }, { when: (s) => s.now >= 2.4, spot: ['ifA'], title: 'The dependency turns slow (5s)', - body: 'Every call now takes 5s. The plain client has no cap — watch in-flight climb without limit as workers block.' }, - { when: (s) => s.mw.rejected > 0, spot: ['poolB'], title: 'The bulkhead holds the line', - body: 'The httpware pool fills to 8 and STOPS admitting more — excess calls fail fast instead of piling up. The client stays responsive for everything else.' }, + body: 'Every call now takes 5s. The plain client has no cap, so its in-flight count keeps climbing as workers block.' }, + { when: (s) => s.mw.rejected > 0, spot: ['poolB'], title: 'The bulkhead is full', + body: 'The httpware pool fills to 8 and admits no more. Further calls fail fast instead of piling up, so the client can still serve other work.' }, { when: (s) => s.now >= 6.0, spot: ['ifA', 'poolB'], title: 'Bounded vs unbounded', - body: 'Plain client: in-flight unbounded, whole client degraded. httpware: in-flight pinned at the pool size, blast radius contained to this one dependency.' }, + body: 'The plain client has no limit on in-flight calls, and the whole client slows down. httpware stays at the pool size, and only calls to this dependency are affected.' }, ], }); }); diff --git a/docs/demos/circuit-breaker.md b/docs/demos/circuit-breaker.md index d526e03..c8c4a49 100644 --- a/docs/demos/circuit-breaker.md +++ b/docs/demos/circuit-breaker.md @@ -1,9 +1,8 @@ -# Circuit Breaker +# Circuit breaker -When a backend goes down, a client without a breaker keeps sending every request -into a slow timeout, piling up in-flight work until it exhausts itself. The breaker -trips after repeated failures and **fast-fails** instead — keeping the client healthy -and probing for recovery. +When a backend goes down, a client without a breaker keeps sending requests that hang +until they time out, and in-flight work piles up. A circuit breaker opens after repeated +failures and fails fast instead, then lets a probe through to check for recovery.
@@ -24,15 +23,15 @@ document.addEventListener('DOMContentLoaded', function () { ], buildStops: () => [ { when: (s) => s.now >= 1.2, spot: ['ifA', 'ifB'], title: 'Two clients, one backend', - body: 'Both are healthy — in-flight near zero on each. The backend is about to die. Keep your eye on these two in-flight counters.' }, - { when: (s) => s.now >= 2.35, spot: ['ifA'], title: 'Backend just went DOWN', - body: 'Every request now hangs ~3s then fails. This plain client keeps sending — watch this number start to climb.' }, - { when: (s) => s.mw.state === 'OPEN', spot: ['brkB', 'ifB'], title: 'The breaker tripped OPEN', - body: '5 failures in a row -> circuit OPEN. It now fast-fails instantly; its in-flight stays flat while the plain client keeps piling up.' }, - { when: (s) => s.now >= 5.6, spot: ['ifA', 'latA', 'ifB', 'latB'], title: 'The gap — this is the point', - body: 'Plain client: in-flight high AND p99 blown to 12s — drowning. Protected client: in-flight flat, p99 still 40ms. Same outage, two outcomes.' }, - { when: (s) => s.mw.recovered, spot: ['brkB'], title: 'Recovery via one probe', - body: 'Backend is back. The breaker admits exactly ONE probe, sees success, and closes — no thundering herd.' }, + body: 'Both are healthy, with almost nothing in flight. The backend is about to fail; watch these two in-flight counters.' }, + { when: (s) => s.now >= 2.35, spot: ['ifA'], title: 'The backend is down', + body: 'Every request now hangs for about 3s and then fails. The plain client keeps sending, so its in-flight count starts to climb.' }, + { when: (s) => s.mw.state === 'OPEN', spot: ['brkB', 'ifB'], title: 'The breaker opens', + body: 'Five failures in a row open the circuit. Requests now fail immediately, so in-flight stays flat while the plain client keeps piling up.' }, + { when: (s) => s.now >= 5.6, spot: ['ifA', 'latA', 'ifB', 'latB'], title: 'Plain vs protected', + body: 'The plain client has a pile of requests in flight and a p99 of 12s. The protected client has a flat in-flight count and a p99 of 40ms.' }, + { when: (s) => s.mw.recovered, spot: ['brkB'], title: 'Recovery through one probe', + body: 'The backend is back. The breaker lets one probe through, sees it succeed, and closes, instead of releasing all the waiting traffic at once.' }, ], }); }); diff --git a/docs/demos/engine.js b/docs/demos/engine.js index b8806c2..241e820 100644 --- a/docs/demos/engine.js +++ b/docs/demos/engine.js @@ -369,9 +369,9 @@ window.HttpwareDemo = (function () { function herdTemplate(config) { const title = config.title || 'Now scale it to 20 clients'; - const intro = config.intro || 'One client retrying a blip is invisible. Twenty clients retrying an ' + - 'outage is a traffic weapon — unless their retries are spread out and capped. ' + - 'These strips show backend call-rate over time. Press play and watch the shape.'; + const intro = config.intro || 'One client retrying a blip goes unnoticed. Twenty clients retrying an ' + + 'outage multiply the load on the backend unless their retries are spread out and capped. ' + + 'These strips show the backend call rate over time. Press play and watch the shape.'; return `

${esc(title)}

@@ -877,7 +877,7 @@ window.HttpwareDemo = (function () { els.play.disabled = false; els.scenLabel.textContent = 'Scenario: ' + selectedScenario.label; els.note.textContent = "Faithful model of httpware's " + describeChain(selectedScenario.chainB) + - ' — not httpware running in your browser.'; + '; httpware itself is not running in your browser.'; brk = null; bulk = null; retryCfg = null; budget = null; budgetExhausted = false; tmoCfg = null; timedOutCount = 0; setupOutageBar(selectedScenario); diff --git a/docs/demos/full-stack.md b/docs/demos/full-stack.md index 81f8fb3..d2f4fb7 100644 --- a/docs/demos/full-stack.md +++ b/docs/demos/full-stack.md @@ -1,8 +1,9 @@ -# Full stack: composing the patterns +# Full stack: combining the patterns -Real clients don't use one pattern — they compose. The recommended order is +Real clients combine these patterns, in the recommended order `AsyncTimeout -> AsyncCircuitBreaker -> AsyncBulkhead -> AsyncRetry -> terminal`. -Here a nasty multi-phase incident hits both clients; watch the layers interlock. +Here both clients go through an incident in three phases, and each layer handles a +different phase.
@@ -26,18 +27,18 @@ document.addEventListener('DOMContentLoaded', function () { ], macroStrip: true, stageLabel: (now) => now < 2 ? 'healthy' - : now < 5 ? 'phase 1 — latency spike' - : now < 8 ? 'phase 2 — brownout' - : now < 12 ? 'phase 3 — hard down' : 'recovered', + : now < 5 ? 'phase 1: latency spike' + : now < 8 ? 'phase 2: brownout' + : now < 12 ? 'phase 3: hard down' : 'recovered', buildStops: () => [ - { when: (s) => s.now >= 2.4, spot: ['poolB', 'elapsedB'], title: 'Phase 1 — latency spike', - body: 'Bulkhead caps concurrency so the slow phase can’t exhaust the client; timeout bounds each operation at 2s.' }, - { when: (s) => s.now >= 5.4, spot: ['ifB'], title: 'Phase 2 — brownout', - body: 'Retry (within budget) recovers many of the transient errors; the budget keeps it from amplifying.' }, - { when: (s) => s.mw.cb && s.mw.cb.state === 'OPEN', spot: ['brkB'], title: 'Phase 3 — hard down', - body: 'Consecutive failures trip the breaker OUTSIDE the retry loop, so it short-circuits the whole retry sequence — one outcome per exhausted sequence, not per attempt.' }, - { when: (s) => s.now >= 13.5, spot: ['ifA', 'latA', 'ifB', 'latB'], title: 'The whole stack vs nothing', - body: 'Plain client: cascading meltdown across every phase. httpware: each layer absorbs the phase it’s built for. That is why they compose.' }, + { when: (s) => s.now >= 2.4, spot: ['poolB', 'elapsedB'], title: 'Phase 1: latency spike', + body: 'The bulkhead caps concurrency so the slow phase cannot exhaust the client, and the timeout ends each operation after 2s.' }, + { when: (s) => s.now >= 5.4, spot: ['ifB'], title: 'Phase 2: brownout', + body: 'Retry recovers many of the transient errors, and the budget stops it from adding much load.' }, + { when: (s) => s.mw.cb && s.mw.cb.state === 'OPEN', spot: ['brkB'], title: 'Phase 3: hard down', + body: 'Repeated failures open the breaker. It sits before the retry layer, so it rejects whole retry sequences and counts one outcome per sequence, not per attempt.' }, + { when: (s) => s.now >= 13.5, spot: ['ifA', 'latA', 'ifB', 'latB'], title: 'Full stack vs none', + body: 'The plain client fails through every phase. In the httpware client, each layer handles the phase it is built for.' }, ], }); }); diff --git a/docs/demos/index.md b/docs/demos/index.md index 0add4a5..ed36b43 100644 --- a/docs/demos/index.md +++ b/docs/demos/index.md @@ -1,15 +1,15 @@ # Resilience demos -Interactive, self-contained walk-throughs of each resilience pattern under load. -Each runs a plain client and an httpware client through the **same** outage, side by -side, and pauses to point out exactly what changes. +Interactive walk-throughs of each resilience pattern under load. Each one runs a +plain client and an httpware client through the same outage side by side, and pauses +to point out what changes. -- [Circuit Breaker](circuit-breaker.md) — stop hammering a dead backend -- [Retry + Budget](retry.md) — rescue blips without causing a storm -- [Bulkhead](bulkhead.md) — contain one slow dependency -- [Timeout](timeout.md) — bound total latency across retries -- [Full stack](full-stack.md) — how they compose +- [Circuit breaker](circuit-breaker.md): stop sending requests to a dead backend. +- [Retry and retry budget](retry.md): recover from blips without causing a retry storm. +- [Bulkhead](bulkhead.md): keep one slow dependency from taking down the client. +- [Timeout](timeout.md): bound total latency across retries. +- [Full stack](full-stack.md): how the patterns combine. !!! note - These are a faithful **model** of httpware's behavior for teaching, not httpware + The demos simulate httpware's behavior for teaching; httpware itself is not running in your browser. See [Resilience](../resilience.md) for the real API. diff --git a/docs/demos/retry.md b/docs/demos/retry.md index ea05156..a62b31f 100644 --- a/docs/demos/retry.md +++ b/docs/demos/retry.md @@ -1,13 +1,12 @@ -# Retry + Retry Budget +# Retry and retry budget -A transient blip should be retried — but a *sustained* outage retried blindly turns -one failure into a retry storm that amplifies the outage. httpware retries with -full-jitter backoff, and a **budget** caps the share of traffic spent on retries so -a dead backend can't be amplified. +A transient blip should be retried, but blindly retrying a sustained outage multiplies +the load on a backend that is already failing. httpware retries with full-jitter +backoff, and a budget caps the share of traffic spent on retries.
-

That was one client recovering from a blip. The real danger of blind retries shows up at scale — when many clients hit a real outage at once.

+

That was one client recovering from a blip. Blind retries do the most harm at scale, when many clients hit a real outage at once.

@@ -22,18 +21,18 @@ document.addEventListener('DOMContentLoaded', function () { budget: { ttl: 10.0, minRetriesPerSec: 10.0, percentCanRetry: 0.2 } } }, ], buildStops: () => [ - { when: (s) => s.now >= 1.2, spot: ['badWrapA', 'badWrapB'], title: 'A backend blip appears', - body: 'Both clients are about to hit the same transient errors. Keep your eye on the ✗ failed counts — they start equal at zero.' }, + { when: (s) => s.now >= 1.2, spot: ['badWrapA', 'badWrapB'], title: 'A backend blip is coming', + body: 'Both clients are about to hit the same transient errors. Watch the ✗ failed counts, which both start at zero.' }, { when: (s) => s.now >= 2.4, spot: ['badWrapA', 'badWrapB'], title: 'Plain surfaces every error; httpware retries', - body: 'The plain client surfaces each error straight to the caller — its ✗ is climbing. httpware retries with backoff, so most of these recover on attempt 2 or 3 and its ✗ barely moves.' }, - { when: (s) => s.now >= 10.0, spot: ['badWrapA', 'badWrapB'], title: 'Blip over — mind the ✗ gap', - body: 'The backend healed. Compare the ✗ failed counts: the plain client surfaced far more failures than httpware. That gap is exactly what retry buys you on a transient blip.' }, + body: 'The plain client passes each error to the caller, so its ✗ count climbs. httpware retries with backoff, most calls succeed on attempt 2 or 3, and its ✗ count barely moves.' }, + { when: (s) => s.now >= 10.0, spot: ['badWrapA', 'badWrapB'], title: 'Blip over: compare the ✗ counts', + body: 'The backend has recovered. The plain client surfaced far more failures than httpware; the difference is what retry saves on a transient blip.' }, ], }); HttpwareDemo.mountHerd('#retry-herd', { clients: 20, - intro: 'Retry rescues a transient blip (above) — but it can’t fix an outage: retried or not, the caller sees the same failures. There the danger isn’t caller failures, it’s amplification. A real backend rarely dies cleanly — it flaps: fails, recovers, fails again. These strips show backend call-rate over time for twenty clients through three dips. Press play and watch the shape.', + intro: 'Retry recovers a transient blip, as above, but not an outage: retried or not, the caller sees the same failures. During an outage the risk is amplification. Real backends rarely fail cleanly; they flap, failing, recovering and failing again. These strips show the backend call rate over time for twenty clients through three dips. Press play and watch the shape.', scenario: { id: 'storm', dur: 12.5, fault: (now) => { const down = (now >= 2.0 && now < 4.0) || (now >= 5.5 && now < 7.5) || (now >= 9.0 && now < 11.0); @@ -44,16 +43,16 @@ document.addEventListener('DOMContentLoaded', function () { buildStops: (sim) => [ { when: (s) => s.revealed >= Math.round(2.2 / sim.dt), spot: ['naiveStrip', 'hwStrip'], title: 'A flapping backend', - body: 'The backend drops, recovers, drops again — three dips (shaded). Every failed request wants to retry. Watch what each herd does to the backend call-rate through the dips.' }, + body: 'The backend goes down, recovers, and goes down again: three dips, shaded. Every failed request wants a retry. Watch what each group of clients does to the backend call rate during the dips.' }, { when: (s) => s.revealed >= Math.round(4.6 / sim.dt), spot: ['naiveStrip'], title: 'Naive: a surge on every dip', - body: 'On each dip, twenty clients retry unbounded — the load piles up into a surge that keeps climbing until the backend recovers, then clears. Three dips, three spikes, with recovery gaps between: the retry storm hitting a backend every time it tries to come back.' }, + body: 'On each dip, twenty clients retry without limit, and the load keeps climbing until the backend recovers. Three dips give three spikes, each hitting the backend just as it tries to come back.' }, { when: (s) => s.revealed >= Math.round(8.2 / sim.dt), spot: ['hwStrip'], title: 'httpware: flat through every dip', - body: 'Full jitter spreads each client’s retries out, and each client’s own max_attempts=3 cap (with its per-client budget as the guarantee at higher volume) limits how much it can add — so httpware holds a low, steady few-times-baseline through every dip instead of spiking.' }, + body: 'Full jitter spreads out the retries of each client, and max_attempts=3 limits how much each client can add, with the per-client budget as the backstop at higher volume. httpware stays at a low, steady few times the baseline through every dip.' }, { when: (s) => s.revealed >= sim.buckets - 1, spot: ['naiveMult', 'hwMult'], - title: 'Peak load: the whole point', - body: 'On its worst dip the naive herd spiked to about 18× the healthy load; httpware never exceeded about 3×, capped by each client’s max_attempts. That flat rate is exactly what lets the backend recover in the gaps — instead of being knocked back down by a retry surge every time it heals.' }, + title: 'Peak load', + body: 'At its worst dip the naive group reached about 18× the healthy load; httpware stayed under about 3×, held there by max_attempts. With the load that low, the backend can recover in the gaps instead of being knocked down again by a retry surge.' }, ], }); }); diff --git a/docs/demos/timeout.md b/docs/demos/timeout.md index 2a30ca4..0d01948 100644 --- a/docs/demos/timeout.md +++ b/docs/demos/timeout.md @@ -1,8 +1,8 @@ # Timeout (total deadline) -`httpx2`'s per-call timeouts bound a single request — but a retry loop with backoff -can still run for many seconds in total. `AsyncTimeout` bounds the **whole** operation, -including every retry and every backoff sleep, so one call can't blow your latency SLA. +`httpx2`'s per-request timeouts bound a single request, but a retry loop with backoff +can still run for many seconds in total. `AsyncTimeout` bounds the whole operation, +every retry and backoff sleep included, so one call can't run past your latency target.
@@ -20,14 +20,14 @@ document.addEventListener('DOMContentLoaded', function () { timeout: { timeout: 2.0 } } }, ], buildStops: () => [ - { when: (s) => s.now >= 1.2, spot: ['ifB', 'elapsedB'], title: 'Retry helps... but costs time', - body: 'httpware retries the brownout. Each retry + backoff adds latency. Watch the elapsed clock on in-flight requests.' }, - { when: (s) => s.now >= 3.0, spot: ['latA', 'elapsedB'], title: 'Unbounded retry = unbounded latency', - body: 'Without a total deadline, a request can churn through every retry and backoff — total wall-clock climbs past any SLA (see the plain lane p99).' }, + { when: (s) => s.now >= 1.2, spot: ['ifB', 'elapsedB'], title: 'Retry helps but costs time', + body: 'httpware retries during the brownout, and each retry and backoff adds latency. Watch the elapsed time on in-flight requests.' }, + { when: (s) => s.now >= 3.0, spot: ['latA', 'elapsedB'], title: 'Unbounded retry, unbounded latency', + body: 'Without a total deadline, a request can go through every retry and backoff, and its total time climbs past any SLA (see the plain lane p99).' }, { when: (s) => s.mw.timedOut > 0, spot: ['elapsedB'], title: 'The deadline fires', - body: 'AsyncTimeout caps the WHOLE operation at 2s. A request that would keep retrying past the deadline is cut off with a bounded TimeoutError — predictable latency, always.' }, + body: 'AsyncTimeout caps the whole operation at 2s. A request that would keep retrying past that is stopped with a TimeoutError.' }, { when: (s) => s.now >= 10.0, spot: ['latA', 'elapsedB'], title: 'Bounded tail latency', - body: 'Retry rescues what it can within budget; the timeout guarantees the tail. Together they bound both failure and latency.' }, + body: 'Retry recovers what it can within its budget, and the timeout caps the tail. Together they limit both failures and latency.' }, ], }); }); diff --git a/docs/dev/contributing.md b/docs/dev/contributing.md index 9d6eeaa..40b7006 100644 --- a/docs/dev/contributing.md +++ b/docs/dev/contributing.md @@ -1,6 +1,6 @@ # Contributing to httpware -Thank you for your interest in contributing. `httpware` is an open-source resilience-first async HTTP client framework for Python, maintained under the [`modern-python`](https://github.com/modern-python) org. +Thank you for your interest in contributing. httpware is maintained under the [`modern-python`](https://github.com/modern-python) org. ## Quick start @@ -14,24 +14,23 @@ just test # pytest with coverage ## Development workflow -1. **Open an issue first** for non-trivial changes — design discussion catches issues earlier than code review. -2. **Branch from `main`**, use a descriptive name (`feat/retry-budget-jitter`, `fix/transport-cancel-leak`). -3. **Run `just lint` and `just test`** locally before pushing. CI will reject changes that fail either. -4. **Add tests** for any code change. Property-based tests (via Hypothesis) are required for concurrency-sensitive code (retry budget, bulkhead, retry interleaving). -5. **Open a pull request** against `main`. PR titles use conventional-commits style (`feat:`, `fix:`, `docs:`, `refactor:`, `test:`, `chore:`). +1. For a non-trivial change, open an issue first so the design can be discussed before code review. +2. Branch from `main` with a descriptive name, such as `feat/retry-budget-jitter` or `fix/transport-cancel-leak`. +3. Run `just lint` and `just test` before pushing. CI rejects changes that fail either. +4. Add tests for every code change. Concurrency-sensitive code (retry budget, bulkhead, retry interleaving) also needs property-based tests with Hypothesis. +5. Open a pull request against `main`. PR titles follow conventional commits (`feat:`, `fix:`, `docs:`, `refactor:`, `test:`, `chore:`). ## Code style - `ruff format` enforces formatting; do not hand-format. - Type-check with `ty` (Astral). Use `# ty: ignore[]` for suppressions, not `# type: ignore`. -- Do NOT use `from __future__ import annotations`. Python 3.11+ is the floor. +- Don't use `from __future__ import annotations`; the minimum Python version is 3.11. - Module, class, and public-method docstrings are required (PEP 257). ## Architecture invariants -These are project invariants, and **enforcement varies — do not assume CI will -catch a violation.** `AGENTS.md` lists which ones are review-only. Do not break -them in pull requests: +Pull requests must keep these rules. CI does not check all of them; `AGENTS.md` +lists the ones only enforced in review. - No `httpx2._*` (private API) usage anywhere in the library. - No `from __future__ import annotations`. @@ -41,7 +40,7 @@ them in pull requests: ## Code of Conduct -By participating in this project, you agree to abide by its Code of Conduct. Be excellent to one another. +By participating in this project, you agree to follow the [`modern-python` Code of Conduct](https://github.com/modern-python/.github/blob/main/CODE_OF_CONDUCT.md). ## License diff --git a/docs/errors.md b/docs/errors.md index fdf7850..2d14230 100644 --- a/docs/errors.md +++ b/docs/errors.md @@ -1,20 +1,18 @@ # Errors reference -`httpware` raises typed exceptions automatically — everything inherits `ClientError`, and HTTP responses with 4xx/5xx status raise status-keyed `StatusError` subclasses without you having to call `response.raise_for_status()`. +Every exception httpware raises subclasses `ClientError`. A response with a 4xx or 5xx status raises a `StatusError` subclass for that status, so you never call `response.raise_for_status()`. -For the resilience-specific errors (`RetryBudgetExhaustedError`, `BulkheadFullError`, `CircuitOpenError`) see the [Resilience reference](resilience.md). - -The status-keyed exception tree is shared between `Client` and `AsyncClient`. Catching `NotFoundError` in sync code uses the same import as catching it in async code (`from httpware import NotFoundError`). +`Client` and `AsyncClient` raise the same classes, so `from httpware import NotFoundError` works for both. The resilience middleware that raise `RetryBudgetExhaustedError`, `BulkheadFullError` and `CircuitOpenError` are described in [Resilience](resilience.md). ## The exception tree ``` ClientError (catch-all for anything httpware raises) -├── TransportError (connection/network/protocol failure pre-response) -│ └── NetworkError (transient — safe to retry; covered by AsyncRetry's defaults) -├── TimeoutError (also inherits builtins.TimeoutError — except OSError catches it) +├── TransportError (connection, network or protocol failure before a response) +│ └── NetworkError (transient and safe to retry; AsyncRetry retries it by default) +├── TimeoutError (also a builtins.TimeoutError, so except OSError catches it) ├── StatusError (got a response but its status was 4xx/5xx) -│ ├── ClientStatusError (any 4xx — fallback for unknown 4xx codes) +│ ├── ClientStatusError (any 4xx; raised as is for unmapped 4xx codes) │ │ ├── BadRequestError (400) │ │ ├── UnauthorizedError (401) │ │ ├── ForbiddenError (403) @@ -22,7 +20,7 @@ ClientError (catch-all for anything httpware raises) │ │ ├── ConflictError (409) │ │ ├── UnprocessableEntityError (422) │ │ └── RateLimitedError (429) -│ └── ServerStatusError (any 5xx — fallback for unknown 5xx codes) +│ └── ServerStatusError (any 5xx; raised as is for unmapped 5xx codes) │ ├── InternalServerError (500) │ └── ServiceUnavailableError (503) ├── RetryBudgetExhaustedError (a retry was needed but the budget refused) @@ -49,9 +47,9 @@ ClientError (catch-all for anything httpware raises) | other 4xx | `ClientStatusError` (fallback) | | other 5xx | `ServerStatusError` (fallback) | -The fallback assumes `400 ≤ status < 600`. Statuses outside that range don't raise (they return the response as-is). +Only statuses from 400 to 599 raise. Any other status returns the response unchanged. -The explicit rows above are also exported as the public `STATUS_TO_EXCEPTION` mapping (`Mapping[int, type[StatusError]]`) — `from httpware import STATUS_TO_EXCEPTION` — so you can look up the class for a status code programmatically (e.g. `STATUS_TO_EXCEPTION.get(404)`). The two fallback rows are not in the mapping; they're applied by the raise logic for any unmapped in-range status. +The numbered rows are exported as `STATUS_TO_EXCEPTION`, a `Mapping[int, type[StatusError]]`, so you can look up a class in code with `STATUS_TO_EXCEPTION.get(404)`. The two fallback rows are not in the mapping. ## Catching strategies @@ -78,28 +76,26 @@ async def fetch(client: AsyncClient, user_id: int) -> dict | None: try: return await client.get(f"/users/{user_id}", response_model=dict) except NotFoundError: - # Specific status — most precise. Convert to None as the "absent" sentinel. + # The most specific catch: treat a missing user as None. return None except StatusError as exc: - # Got a response, but its status was 4xx/5xx and not one we handle specifically. - # exc.response.* is available — headers, content, request, etc. + # Any other 4xx or 5xx. exc.response has the headers, body and request. _LOGGER.warning("upstream returned %s for %s", exc.response.status_code, exc.response.request.url) raise except NetworkError: - # Transient transport failure. Already retried by the default AsyncRetry middleware - # (if installed) when the method was idempotent. Seeing this means retries - # exhausted or the method was non-idempotent. + # Transient transport failure. If the client has AsyncRetry, an idempotent + # request was already retried and this is the last attempt's error. raise except (RetryBudgetExhaustedError, BulkheadFullError) as exc: - # Resilience refusal — backpressure signal. Back off the caller. + # A resilience middleware refused the call; ease off upstream. _LOGGER.error("resilience refused: %s", exc) raise except ClientError: - # Catch-all for anything else httpware raised. + # Anything else httpware raised. raise ``` -`TimeoutError` is doubly-inherited: `except builtins.TimeoutError` and `except OSError` both catch it (matches what `asyncio.wait_for` raises). This lets stdlib-style timeout handling Just Work. +`httpware.TimeoutError` also subclasses `builtins.TimeoutError`, the class `asyncio.wait_for` raises, so `except builtins.TimeoutError` and `except OSError` both catch it. ## `exc.response.*` access pattern @@ -107,7 +103,7 @@ For any `StatusError` subclass, the raw `httpx2.Response` is on `exc.response`: ```python exc.response.status_code # 404 -exc.response.headers # httpx2.Headers — case-insensitive +exc.response.headers # httpx2.Headers, case-insensitive exc.response.content # raw bytes exc.response.text # decoded body exc.response.json() # parsed JSON (raises if not JSON) @@ -116,23 +112,23 @@ exc.response.request.url # the failing URL (httpx2.URL) exc.response.request.method # the HTTP method ``` -**Security note:** `__repr__` and the exception's summary message strip `user:pass@` userinfo and mask the values of known-sensitive query and URL-fragment parameters (`api_key`, `apikey`, `access_token`, `refresh_token`, `token`, `secret`, `client_secret`, `password`, `passwd`, `pwd`, `auth`, `authorization`, `sig`, `signature`, `key`, `private_key`, `session`, `sessionid`, `x-api-key`) as `REDACTED`, preserving the keys. Query values under other names are **not** masked, so still avoid putting non-standard secrets in query strings. Note that request *headers* (`Authorization`, `Cookie`, etc.) are never redacted — see `exc.response.request.headers` above. +The exception's `repr` and message strip `user:pass@` userinfo and mask the values of known-sensitive query and URL-fragment parameters (`api_key`, `apikey`, `access_token`, `refresh_token`, `token`, `secret`, `client_secret`, `password`, `passwd`, `pwd`, `auth`, `authorization`, `sig`, `signature`, `key`, `private_key`, `session`, `sessionid`, `x-api-key`) as `REDACTED`, keeping the keys. Values under other names are left as they are, so avoid putting other secrets in query strings. Request headers such as `Authorization` and `Cookie` are never redacted, and `exc.response.request.headers` returns them in full. ## Resilience-error payloads `RetryBudgetExhaustedError` carries: -- `last_response: httpx2.Response | None` — the last response observed before the budget refused (None if all failures were transport-level) -- `last_exception: BaseException | None` — the last exception observed before the budget refused -- `attempts: int` — number of attempts already completed +- `last_response: httpx2.Response | None`: the last response before the budget refused, or `None` if every failure was at the transport level. +- `last_exception: BaseException | None`: the last exception before the budget refused. +- `attempts: int`: how many attempts were completed. `BulkheadFullError` carries: -- `max_concurrent: int` — the configured cap -- `acquire_timeout: float | None` — the configured timeout +- `max_concurrent: int`: the configured cap. +- `acquire_timeout: float | None`: the configured timeout. `CircuitOpenError` carries: -- `retry_after: float | None` — seconds until the circuit will next admit a probe; `None` when a concurrent probe is already in flight (HALF_OPEN slot taken). +- `retry_after: float | None`: seconds until the circuit admits its next probe, or `None` when a probe is already in flight. -Use these for caller-side logging / alerting: +You can use these fields in your own logging and alerts: ```python except RetryBudgetExhaustedError as exc: @@ -145,13 +141,13 @@ except RetryBudgetExhaustedError as exc: ## `DecodeError` -`DecodeError` is raised when `response_model=` is set on a request and the active `ResponseDecoder` failed to parse the response body. The HTTP call itself succeeded — status was 2xx/3xx and the transport delivered the body intact — but the body could not be coerced into the requested model. The exception is raised independently of which decoder is in use (`PydanticDecoder`, `MsgspecDecoder`, or a third-party decoder), so `except httpware.ClientError` is sufficient to cover the response-model decode path. +`DecodeError` means the HTTP call succeeded but the decoder could not turn the body into the `response_model=` type. Every decoder's failures are wrapped this way, whether it is `PydanticDecoder`, `MsgspecDecoder` or your own, so `except httpware.ClientError` covers decoding too. Fields: -- `response: httpx2.Response` — the response whose body failed to decode. Status, headers, and the originating `request` are all available via `exc.response.*`. -- `model: type` — the type that was passed as `response_model=`. -- `original: BaseException` — the underlying library exception (e.g., `pydantic.ValidationError`, `msgspec.ValidationError`, `msgspec.DecodeError`). Also available via `exc.__cause__`. +- `response: httpx2.Response`: the response whose body failed to decode, with its status, headers and `request`. +- `model: type`: the type passed as `response_model=`. +- `original: BaseException`: the decoder's own exception, such as `pydantic.ValidationError`, `msgspec.ValidationError` or `msgspec.DecodeError`. It is also `exc.__cause__`. ```python from httpware import AsyncClient, DecodeError @@ -171,37 +167,37 @@ except DecodeError as exc: ## `MissingDecoderError` -Raised by `send()` / `send_with_response()` / verb methods when `response_model=` is set but no registered decoder claims the model. Carries: +`send()`, `send_with_response()` and the verb methods raise `MissingDecoderError` when `response_model=` is set and no registered decoder accepts the type. It carries: -- `model: type` — the `response_model=` value that wasn't claimed. -- `registered_names: tuple[str, ...]` — class names of the registered decoders that all rejected the model. Empty tuple means no decoders were registered. +- `model: type`: the `response_model=` value nobody accepted. +- `registered_names: tuple[str, ...]`: class names of the decoders that rejected it. An empty tuple means no decoders were registered. -The message reads `no decoder for response_model=: `, and the corrective action depends on the hint. The two hints, verbatim: +The message reads `no decoder for response_model=: `. There are two hints. -- **No decoders were registered** — install an extra or pass an explicit decoder list: +- No decoders were registered. Install an extra or pass a decoder list: no decoders registered. Install `pip install httpware[pydantic]` or `pip install httpware[msgspec]`, or pass decoders=[...] explicitly. -- **Registered decoders all rejected the model** — your `response_model` type is exotic enough that neither built-in claims it; pass a custom `ResponseDecoder` via `decoders=[...]`: +- Every registered decoder rejected the type. Neither built-in handles it, so pass your own `ResponseDecoder` in `decoders=[...]`: registered decoders (PydanticDecoder + MsgspecDecoder) all rejected it. Pass a custom decoder via decoders=[...]. -Unlike `DecodeError`, this error fires *before* the HTTP request — no traffic is sent. +Unlike `DecodeError`, this error is raised before the request is sent. ## `ResponseTooLargeError` -Both `Client` and `AsyncClient` accept a `max_response_body_bytes: int | None = None` constructor argument. It's an opt-in cap — the default `None` means unbounded, matching current behavior. When set, a response body that exceeds the cap raises `ResponseTooLargeError` instead of being returned. The check is status-agnostic (a `200` can trip it just as easily as a `4xx`/`5xx`), and it counts **decoded** bytes. It fires from the non-streaming terminal (`send()` / verb methods) and from `stream()`'s internal error pre-read; bytes you pull yourself via `stream()` iteration are never capped. Combining the cap with `follow_redirects=True`, on the client or on a passed `httpx2_client`, raises `ValueError`: `httpx2` reads every intermediate redirect body without it. +Both clients accept `max_response_body_bytes: int | None = None`. By default there is no limit. When it is set, a response body larger than the cap raises `ResponseTooLargeError` instead of being returned, whatever the status: a `200` trips it as easily as a `500`. The cap counts decoded bytes, after decompression. It applies to `send()` and the verb methods, and to the error body that `stream()` reads before raising a `StatusError`. Bytes you read yourself while iterating a `stream()` are never capped. Setting the cap together with `follow_redirects=True`, on the client or on a passed `httpx2_client`, raises `ValueError`, because `httpx2` reads every intermediate redirect body without the cap. `ResponseTooLargeError` carries: -- `status_code: int` — the response's HTTP status code. -- `limit: int` — the configured `max_response_body_bytes` value that was exceeded. -- `content_length: int | None` — the server-declared `Content-Length`, when known. -- `reason: Literal["declared", "streamed"]` — which trip mode fired: - - `"declared"` — the declared `Content-Length` already exceeded `limit`; the body was rejected before any byte was read, and `content_length` holds the offending value. - - `"streamed"` — the decoded body crossed `limit` mid-read (the chunked-transfer or compression-bomb case); the true oversized length is unknown by design, so `content_length` is whatever (possibly absent or understated) value the server declared. +- `status_code: int`: the response's HTTP status code. +- `limit: int`: the `max_response_body_bytes` value that was exceeded. +- `content_length: int | None`: the `Content-Length` the server declared, if any. +- `reason: Literal["declared", "streamed"]`: why the cap tripped. + - `"declared"`: the declared `Content-Length` was already over `limit`, so no body was read. `content_length` holds that value. + - `"streamed"`: the decoded body passed `limit` while being read, as with chunked transfer or a compression bomb. Reading stops there, so the real size is unknown and `content_length` is only what the server declared, which may be missing or too small. -It is a non-status `ClientError` — it does not carry a `StatusError`-style positional `response` and is not in `STATUS_TO_EXCEPTION`. Because it's neither a `StatusError`, `NetworkError`, nor `TimeoutError`, it is not retried by `AsyncRetry` and does not count toward the circuit breaker. +`ResponseTooLargeError` is a `ClientError` but not a `StatusError`: it has no `response` and is not in `STATUS_TO_EXCEPTION`. `AsyncRetry` does not retry it and the circuit breaker does not count it. ```python from httpware import AsyncClient, ResponseTooLargeError @@ -217,6 +213,6 @@ except ResponseTooLargeError as exc: ## See also -- **[Resilience reference](resilience.md)** — `AsyncRetry`, `RetryBudget`, `AsyncBulkhead` parameter tables. -- **[Middleware guide](middleware.md)** — the `@async_on_error` decorator can translate exceptions into responses. -- **`src/httpware/errors.py`** — the tree itself; the construction rules for both halves of it are enforced in `tests/test_errors.py`. +- [Resilience](resilience.md): the middleware that raise `RetryBudgetExhaustedError`, `BulkheadFullError` and `CircuitOpenError`. +- [Middleware](middleware.md): the `@async_on_error` decorator can turn exceptions into responses. +- [`src/httpware/errors.py`](https://github.com/modern-python/httpware/blob/main/src/httpware/errors.py): the exception classes. diff --git a/docs/index.md b/docs/index.md index 28b157c..4f5b742 100644 --- a/docs/index.md +++ b/docs/index.md @@ -7,15 +7,7 @@
-A Python HTTP client framework with sync and async clients for building resilient service clients. `httpware` is a thin opinionated wrapper around `httpx2` — it re-exports `httpx2.Request`/`httpx2.Response` as the public request/response surface, adds a middleware chain (with a built-in resilience suite: `AsyncRetry`/`Retry` + `RetryBudget`, `AsyncBulkhead`/`Bulkhead`, `AsyncCircuitBreaker`/`CircuitBreaker`, and `AsyncTimeout`), opt-in typed response decoding, and a status-keyed exception tree raised automatically on 4xx/5xx. - -## Why httpware - -Typed exceptions per HTTP status, typed response bodies, and composable -resilience (retry, bulkhead, circuit breaker, timeout) — a thin wrapper over -`httpx2`, not a new HTTP abstraction. See the -[project README](https://github.com/modern-python/httpware#why-httpware) for -the full pitch. +httpware wraps `httpx2` to give you sync and async clients for calling other services. It adds a middleware chain with built-in retry, bulkhead, circuit breaker and timeout, optional typed response decoding, and an exception per HTTP status raised automatically on 4xx and 5xx. Requests and responses are plain `httpx2.Request` and `httpx2.Response` objects. > **Status:** Pre-1.0. Public API is subject to change between minor releases until v1.0. @@ -28,14 +20,16 @@ pip install httpware Optional extras: ```bash -pip install httpware[pydantic] # PydanticDecoder — handles BaseModel + dataclasses + primitives + generics -pip install httpware[msgspec] # MsgspecDecoder — handles Struct + dataclasses + primitives + generics -pip install httpware[pydantic,msgspec] # both extras — both decoders register; BaseModel routes to pydantic, Struct to msgspec +pip install httpware[pydantic] # PydanticDecoder: BaseModel, dataclasses, primitives, generics +pip install httpware[msgspec] # MsgspecDecoder: Struct, dataclasses, primitives, generics +pip install httpware[pydantic,msgspec] # both; BaseModel goes to pydantic, Struct to msgspec +pip install httpware[otel] # OpenTelemetry span events +pip install httpware[all] # pydantic, msgspec, and otel ``` ## First request -**Async usage:** +Async: ```python import asyncio @@ -52,7 +46,7 @@ async def main() -> None: asyncio.run(main()) ``` -**Sync usage:** +Sync: ```python from httpware import Client @@ -62,24 +56,9 @@ with Client(base_url="https://jsonplaceholder.typicode.com") as client: print(response.json()) ``` -`base_url` must not contain a query string: constructing a client with one raises `ValueError`. Put query parameters shared by every request in `params=` instead. +### Typed responses -Every other keyword of `httpx2.AsyncClient`/`httpx2.Client` (`verify`, `proxy`, `http2`, `transport`, `follow_redirects`, ...) is forwarded to the `httpx2` client that `httpware` builds and closes. Two are refused: `cert`, deprecated by `httpx2` in favour of an `ssl.SSLContext` passed as `verify`, and `event_hooks`, which run below the middleware chain; use [middleware](middleware.md) instead. - -```python -import ssl - -from httpware import AsyncClient - -client = AsyncClient( - base_url="https://internal.example", - verify=ssl.create_default_context(cafile="/etc/ssl/internal-ca.pem"), -) -``` - -To share one connection pool between several clients, build the `httpx2` client yourself and pass it as `httpx2_client=`. It is then yours to close, and none of the options above can be combined with it. - -Typed decoding via `response_model=` works the same way in both worlds: +Pass `response_model=` to get a decoded body instead of the response. It works the same way on both clients: ```python from httpware import AsyncClient @@ -97,20 +76,13 @@ async def main() -> None: print(user.name) ``` -Need the raw response **and** a decoded body from the same call (e.g., for header-based pagination)? See [Link header pagination](recipes/link-header-pagination.md) — it uses `send_with_response`. +The client tries its `decoders` in order and uses the first one whose `can_decode` returns `True`, so list order decides which decoder wins when more than one could handle a type. If none can, the call raises `MissingDecoderError` before the request is sent. See [Decoders](decoders.md) for the resolution rules and how pydantic and msgspec types are routed. -### Decoder dispatch - -When `response_model=` is set, the client walks `decoders` in order and picks -the first decoder whose `can_decode` returns `True`; ordering encodes your -preference for shapes more than one decoder could claim. If none claims your -`response_model`, the call raises `MissingDecoderError` *before* the HTTP -request. See **[Decoders](decoders.md)** for the resolution rules and -pydantic/msgspec routing. +To get the raw response and the decoded body from the same call, for example to read pagination headers, use `send_with_response`. The [Link header pagination](recipes/link-header-pagination.md) recipe shows it in a loop. ### With resilience middleware -Compose resilience middleware at construction; `AsyncBulkhead` goes outside `AsyncRetry` so one slot covers all retry attempts. +Pass resilience middleware when you build the client. Put `AsyncBulkhead` before `AsyncRetry` so one slot covers all retry attempts of a call. ```python from httpware import AsyncClient, AsyncBulkhead, AsyncRetry @@ -127,9 +99,11 @@ async def main() -> None: user = await client.get("/users/1", response_model=User) ``` +[Resilience](resilience.md) has the full recommended order, including the circuit breaker and timeout. + ### Streaming responses -For large responses or server-sent events, stream the body chunk-by-chunk. `stream()` is an async context manager: +For large responses or server-sent events, stream the body in chunks. `stream()` is an async context manager: ```python from httpware import AsyncClient @@ -142,39 +116,52 @@ async def main() -> None: process(chunk) ``` -`stream()` auto-raises `StatusError` subclasses on 4xx/5xx with the response body pre-read, so `exc.response.content` is accessible from the caught exception. +`stream()` raises `StatusError` subclasses on 4xx and 5xx like every other call. It reads the error body first, so `exc.response.content` is available on the caught exception. -It does NOT pass through the middleware chain: `AsyncRetry`, `AsyncBulkhead`, and any custom middleware are bypassed. (AsyncRetry separately refuses to retry any request — stream or non-stream — whose body was an async-iterable, since streams can't replay across attempts.) +`stream()` does not go through the middleware chain: `AsyncRetry`, `AsyncBulkhead`, and your own middleware are all skipped. Separately, `AsyncRetry` never retries a request whose body was an async iterable, streamed or not, because the iterable cannot be replayed. -### Capping response body size +## Client options + +`base_url` must not contain a query string; a client built with one raises `ValueError`. Put query parameters shared by every request in `params=` instead. + +The other keywords of `httpx2.AsyncClient` and `httpx2.Client` (`verify`, `proxy`, `http2`, `transport`, `follow_redirects`, ...) are passed through to the `httpx2` client that httpware builds and closes. Two are refused. `cert` is deprecated by `httpx2` in favour of an `ssl.SSLContext` passed as `verify`. `event_hooks` run below the middleware chain, so use [middleware](middleware.md) instead. + +```python +import ssl + +from httpware import AsyncClient -Both clients accept an opt-in `max_response_body_bytes: int | None = None`. When set, a response body that exceeds the cap raises `ResponseTooLargeError` instead of being returned; the default `None` is unbounded. See **[Errors](errors.md#responsetoolargeerror)** for the full trip conditions. +client = AsyncClient( + base_url="https://internal.example", + verify=ssl.create_default_context(cafile="/etc/ssl/internal-ca.pem"), +) +``` + +To share one connection pool between several clients, build the `httpx2` client yourself and pass it as `httpx2_client=`. You then close it yourself, and you cannot combine it with any of the options above. + +### Capping response body size -## Errors +Both clients accept `max_response_body_bytes: int | None = None`. When it is set, a response body larger than the cap raises `ResponseTooLargeError` instead of being returned. The default, `None`, sets no limit. [Errors](errors.md#responsetoolargeerror) lists exactly when the cap applies. -All errors inherit `httpware.ClientError`: 4xx/5xx responses raise a typed -`StatusError` subclass automatically, and `response_model=` decode failures -raise `DecodeError`. See **[Errors](errors.md)** for the full tree and -catching strategies. +## Errors and observability -## Observability +Every exception httpware raises subclasses `httpware.ClientError`. A 4xx or 5xx response raises a `StatusError` subclass, and a body that fails `response_model=` decoding raises `DecodeError`. [Errors](errors.md) has the full tree. -Every resilience middleware emits stdlib-`logging` records (always) and OTel -span events (when `opentelemetry-api` is installed), under stable logger and -event names. See **[Observability](observability.md)** for the full contract. +The resilience middleware log through stdlib `logging` and, when `opentelemetry-api` is installed, add span events. Logger and event names are stable; [Observability](observability.md) lists them. ## Where to go next -- **[Resilience reference](resilience.md)** — every parameter on `AsyncRetry`, `RetryBudget`, and `AsyncBulkhead`; the retry-rule matrix; Retry-After parsing; budget sharing. -- **[Middleware guide](middleware.md)** — write your own middleware. Covers the AsyncMiddleware Protocol, the phase decorators, a worked Request-ID propagation example, and OpenTelemetry wiring. -- **[Errors reference](errors.md)** — the full exception tree, catching strategies, `exc.response.*` access pattern. -- **[Observability](observability.md)** — the stdlib-`logging` and OTel span-event contract emitted by the resilience middleware. -- **[Testing guide](testing.md)** — mock-transport injection pattern for testing code that uses `httpware`. -- **[Recipes](recipes/modern-di.md)** — wiring `AsyncClient` into a `modern-di` container. -- **[Decision records](https://github.com/modern-python/httpware/tree/main/docs/adr)** — the alternatives that were considered and rejected, and why. -- **[Contributing](dev/contributing.md)** — setup, conventions, workflow. -- **[Release notes](https://github.com/modern-python/httpware/releases)** — per-version changelogs. +- [Resilience](resilience.md): every parameter of retry, retry budget, bulkhead, circuit breaker and timeout, plus the order to compose them in. +- [Middleware](middleware.md): the middleware protocol, phase decorators, and a request-ID example. +- [Errors](errors.md): the exception tree, catching strategies, and what each exception carries. +- [Decoders](decoders.md): how `response_model=` picks a decoder, and how to write your own. +- [Observability](observability.md): logger and event names, and OpenTelemetry wiring. +- [Testing](testing.md): testing code that uses httpware with `httpx2.MockTransport`. +- [Recipes](recipes/modern-di.md): `modern-di` wiring, phase decorator patterns, Link header pagination. +- [Decision records](https://github.com/modern-python/httpware/tree/main/docs/adr): alternatives that were considered and rejected, and why. +- [Contributing](dev/contributing.md): setup, conventions, workflow. +- [Release notes](https://github.com/modern-python/httpware/releases): changes in each version. ## Part of `modern-python` -`httpware` ships under the [`modern-python`](https://github.com/modern-python) org. See the org profile for the categorized index of related templates and libraries. +httpware is part of the [`modern-python`](https://github.com/modern-python) org. The org profile has the categorized index of related templates and libraries. diff --git a/docs/middleware.md b/docs/middleware.md index df06e0f..bacbe8e 100644 --- a/docs/middleware.md +++ b/docs/middleware.md @@ -1,19 +1,19 @@ # Middleware -`httpware`'s primary extension point is the **AsyncMiddleware protocol**. Middleware lets you add cross-cutting behavior — request-ID propagation, auth header injection, structured tracing, custom resilience policies, anything that wraps "send a request, get a response" — without subclassing `AsyncClient` or touching the transport. +Middleware is the main way to extend httpware. A middleware wraps every call a client makes, so it can add request IDs or auth headers, record traces, or apply your own resilience policy without subclassing `AsyncClient` or touching the transport. -The built-in `AsyncRetry` and `AsyncBulkhead` middleware are themselves implementations of this protocol; nothing about them is privileged. If you want a circuit breaker, a rate limiter, or a header-injecting auth layer, write a middleware. +The built-in retry, bulkhead, circuit breaker and timeout are ordinary middleware written against the same protocol. A rate limiter or an auth layer you write yourself plugs in the same way. ## Choosing where behavior lives -Middleware is for *cross-cutting* concerns — behavior that should apply to every call through a client. For everything else, reach for a more specific tool: +Middleware is for behavior that applies to every call through a client. Other cases have a better place: -- **Per-call behavior that doesn't apply to other calls:** pass it through `request.extensions=` (or the `extensions=` kwarg at the call site) instead of a middleware. -- **Instance state or two-sided inspection** (a counter, a CircuitBreaker's open/closed flag, timing that needs both the request and its response, or interleaving behavior around the `await next(...)` call): write a raw `AsyncMiddleware`/`Middleware` class rather than a phase decorator — decorators are a convenience for the cases where a single function suffices. -- **Transform that doesn't need `httpware`'s exception mapping or chain ordering** (pure request/response side effects at the lowest level, including post-redirect hops): use `event_hooks` on an `httpx2` client you build yourself and pass as `httpx2_client=`. Phase decorators and middleware participate in the `httpware` chain (they see `httpware` exceptions and compose with `AsyncRetry`/`AsyncBulkhead`); `event_hooks` run a layer below, on every transport attempt. That is also why `httpware` refuses `event_hooks=` as a client option: an `httpx2.HTTPStatusError` raised in a hook (say, by `raise_for_status()`) reaches the chain as a plain `TransportError` instead of a `StatusError`, so `AsyncRetry` never retries it and the circuit breaker never counts it. -- **URL or header validation:** `httpx2` owns it — don't reimplement. -- **HTTP-level span creation for tracing:** install `opentelemetry-instrumentation-httpx` instead of writing an OTel middleware in httpware. `opentelemetry-instrumentation-httpx` already covers transport-level tracing, so a separate httpware layer would duplicate it. See [Observability](observability.md). -- **Redaction:** httpware redacts URLs before they reach logs, telemetry, and error messages — `user:pass@` userinfo is stripped and sensitive query- and fragment-parameter values are masked (`_internal/redaction.py`). It does **not** inspect or redact headers or request/response bodies, so if your own middleware logs those, redact them yourself (e.g. with a `logging.Filter`). +- Behavior for a single call: pass it through `request.extensions=` (or the `extensions=` kwarg at the call site) instead of a middleware. +- Instance state or logic on both sides of the call (a counter, a circuit breaker's open/closed flag, timing that needs both the request and its response): write an `AsyncMiddleware` or `Middleware` class. Phase decorators only cover logic that fits in one function. +- Side effects that don't need httpware's exceptions or chain order, including on post-redirect hops: use `event_hooks` on an `httpx2` client you build yourself and pass as `httpx2_client=`. Middleware and phase decorators run in the httpware chain, see httpware exceptions, and compose with `AsyncRetry` and `AsyncBulkhead`. `event_hooks` run a layer below, on every transport attempt. That is also why `httpware` refuses `event_hooks=` as a client option: an `httpx2.HTTPStatusError` raised in a hook (say, by `raise_for_status()`) reaches the chain as a plain `TransportError` instead of a `StatusError`, so `AsyncRetry` never retries it and the circuit breaker never counts it. +- URL or header validation: `httpx2` already does it. +- Creating HTTP spans for tracing: install `opentelemetry-instrumentation-httpx`, which already traces every transport call. See [Observability](observability.md). +- Redaction: httpware redacts URLs before they reach logs, telemetry, and error messages. It strips `user:pass@` userinfo and masks the values of sensitive query and fragment parameters ([Errors](errors.md#excresponse-access-pattern) lists them). It does not touch headers or bodies, so if your middleware logs those, redact them yourself, for example with a `logging.Filter`. ## Writing your own @@ -38,17 +38,17 @@ The chain is composed once at `AsyncClient.__init__` and frozen for the client's Calling `await next(request)` forwards to the next layer (or, eventually, to the terminal that hits `httpx2`). You can: -- **Forward unchanged:** `return await next(request)` -- **Modify the request first:** mutate `request.headers` (or build a replacement) before forwarding -- **Inspect or replace the response:** call `await next(...)`, then act on what comes back -- **Short-circuit:** return a synthesized `httpx2.Response` without calling `next` at all -- **Wrap the call in error handling:** `try: return await next(...) except ...` to translate failures +- forward it unchanged with `return await next(request)` +- change `request.headers`, or build a new request, before forwarding +- call `await next(...)` and inspect or replace the response +- return your own `httpx2.Response` without calling `next` +- wrap `await next(...)` in `try`/`except` to translate failures -Whatever you do, return an `httpx2.Response`. Raising an exception propagates up the chain (AsyncRetry catches retryable exceptions; everything else surfaces to the caller). +Either return an `httpx2.Response` or raise. An exception travels up the chain: `AsyncRetry` catches the ones it retries, and the rest reach the caller. ### Phase decorators -For the common cases where you don't need state-keeping on `self` and don't need to wrap the full `await next(...)` call, `httpware.middleware` exports three decorators that turn a single async function into an `AsyncMiddleware`: +When you need no state on `self` and don't need to wrap `await next(...)`, three decorators turn a single async function into an `AsyncMiddleware`: ```python from httpware import async_before_request, async_after_response, async_on_error @@ -60,11 +60,11 @@ from httpware import async_before_request, async_after_response, async_on_error | `@async_after_response` | `async (request, response) -> response` | Transform the incoming response (decode, log, attach metadata). | | `@async_on_error` | `async (request, exc) -> response \| None` | Translate or absorb a failure. Return `None` to re-raise. Catches `Exception` (not `BaseException`), so `asyncio.CancelledError` propagates. | -See the **[Phase decorator recipes](recipes/phase-decorator-patterns.md)** for worked examples covering each decorator: bearer-token injection, correlation-ID propagation from `contextvars`, status-class counter, and `NetworkError` fallback. +The [phase decorator recipes](recipes/phase-decorator-patterns.md) have an example for each decorator: bearer-token injection, a correlation ID from `contextvars`, a status-class counter, and a `NetworkError` fallback. ### Worked example: request-ID propagation -A `RequestIdMiddleware` that assigns a per-call UUID, injects it as an outgoing header, and logs it alongside the response status. This is the canonical "trace every request through your distributed system" pattern. +This `RequestIdMiddleware` gives each call a UUID, sends it as a header, and logs it with the response status, so you can follow one request across services. ```python import logging @@ -80,12 +80,9 @@ _LOGGER = logging.getLogger("myapp.request_id") class RequestIdMiddleware: - """Assign a per-call X-Request-Id; log it on response. + """Assign a per-call X-Request-Id and log it with the response status. - Place OUTSIDE AsyncRetry so all attempts of the same call share one ID - (so a single call's retries all surface under the same correlation - key in your logs, and match the URL attribute on httpware.retry's - emitted events). + Place it before AsyncRetry so every attempt of one call shares the ID. """ def __init__(self, *, header: str = "X-Request-Id") -> None: @@ -110,13 +107,13 @@ async def main() -> None: await client.get("/users/1") ``` -A note on logger names: the example logs under `myapp.request_id`, NOT under `httpware.*`. The `httpware.*` namespace is reserved for events emitted by the library itself (see [Observability](observability.md) — `httpware.retry`, `httpware.bulkhead`, `httpware.circuit_breaker`, and `httpware.timeout` are stable contracts). Consumer middleware should use your application's own logger namespace. +The example logs under `myapp.request_id`. Keep your middleware's logs in your application's namespace: `httpware.*` is reserved for the library's own loggers, whose names are a stable contract (see [Observability](observability.md)). -The example pairs naturally with the 0.6.0 observability events: a `httpware.retry` `retry.giving_up` log record carries a `url` attribute, and your `RequestIdMiddleware` set an `X-Request-Id` for that same call. Correlate the two in your log aggregator and you have end-to-end visibility from "this user's request" to "we gave up after N retries." +A `retry.giving_up` record from `httpware.retry` carries the call's `url`, and this middleware logged an `X-Request-Id` for the same call. Joining the two in your log aggregator tells you which request gave up after its retries. ### Enriching the active span -See **[Wiring OpenTelemetry](observability.md#wiring-opentelemetry)** for how to wire the OTel SDK and `opentelemetry-instrumentation-httpx` so `httpware` HTTP calls get a span at all. Once a span is active, your own middleware can attach to it the same way `httpware`'s built-in resilience middleware does — no additional setup needed: +httpware calls only get a span once you set up the OTel SDK and `opentelemetry-instrumentation-httpx`; [Wiring OpenTelemetry](observability.md#wiring-opentelemetry) shows how. With a span active, your middleware can add to it the same way the built-in resilience middleware do: ```python import httpx2 @@ -132,17 +129,17 @@ class SpanEnrichingMiddleware: return response ``` -When no span is active, `get_current_span()` returns a `NonRecordingSpan` whose `set_attribute`/`add_event` are documented no-ops, so this is safe to call unconditionally. +When no span is active, `get_current_span()` returns a `NonRecordingSpan` whose `set_attribute` and `add_event` do nothing, so the call is always safe. ### Sync middleware -The same protocol shape, sync flavor. Use these when wiring middleware into a sync `Client` instead of `AsyncClient`. +A sync `Client` takes sync middleware, which has the same shape without `async`: ```python from httpware import Middleware, Next, before_request, after_response, on_error ``` -A sync `Middleware` is a structural protocol — any callable with the right signature satisfies it: +`Middleware` is a structural protocol, so any callable with the right signature works: ```python import logging @@ -168,7 +165,7 @@ with Client(base_url="https://api.example.com", middleware=[LoggingMiddleware()] client.get("/users/1") ``` -Phase decorators (`@before_request`, `@after_response`, `@on_error`) have the same semantics as their `@async_*` siblings, but wrap sync functions: +`@before_request`, `@after_response` and `@on_error` behave like their `@async_*` versions but wrap sync functions: ```python import uuid @@ -192,9 +189,9 @@ with Client(base_url="https://api.example.com", middleware=[add_request_id]) as client.get("/users/1") ``` -Sync and async middleware classes do not interop: a `Middleware` cannot be passed to `AsyncClient(middleware=...)` and vice versa. Pick the flavor matching your client. +Sync and async middleware don't mix: pass `Middleware` to `Client` and `AsyncMiddleware` to `AsyncClient`. ## See also -- **`src/httpware/middleware/resilience/`** — `AsyncRetry`, `AsyncBulkhead`, `RetryBudget` as real-world consumers of this exact protocol. -- **[Quick-Start composition example](index.md#with-resilience-middleware)** — composing built-in middleware. +- [Resilience](resilience.md): the built-in middleware and the order to compose them in. +- [`src/httpware/middleware/resilience/`](https://github.com/modern-python/httpware/tree/main/src/httpware/middleware/resilience): the built-in middleware's source, written against this same protocol. diff --git a/docs/observability.md b/docs/observability.md index 30a36d8..7d532f6 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -1,10 +1,8 @@ # Observability -This page is the stable reference for the logger names, event names, and OpenTelemetry wiring that the resilience middleware emit. +The resilience middleware report what they do in two ways: as stdlib `logging` records, always, and as OpenTelemetry span events when `opentelemetry-api` is installed. Sync and async classes emit the same event names and payloads, so one dashboard covers both. -All resilience middleware emit operational events via two channels — stdlib `logging` records (always on) and OpenTelemetry span events (when `opentelemetry-api` is installed). Event names and payloads are identical across sync and async; dashboards built against one class apply unchanged to the other. - -Logger names and event names are the stable public contract: +The logger and event names below are a stable public contract: | Logger | Events | |---|---| @@ -13,34 +11,31 @@ Logger names and event names are the stable public contract: | `httpware.circuit_breaker` | `circuit.opened` (WARNING), `circuit.rejected` (WARNING), `circuit.half_open` (INFO), `circuit.closed` (INFO) | | `httpware.timeout` | `timeout.exceeded` (WARNING) | -Each log record carries an `event` field with the event-name string (e.g. `event="circuit.opened"`), usable for log-aggregator filtering. Events from `AsyncKeyedCircuitBreaker` / `KeyedCircuitBreaker` also carry `circuit_key`, naming the circuit they belong to. See [resilience.md](resilience.md) for the full event tables per middleware. +Each log record has an `event` field holding the event name, such as `event="circuit.opened"`, which you can filter on in your log aggregator. Events from `AsyncKeyedCircuitBreaker` and `KeyedCircuitBreaker` also carry `circuit_key`, the circuit they belong to. [Resilience](resilience.md) describes when each event fires. ```python import logging -# Enable visibility into resilience operational events +# Show the resilience middleware's events logging.getLogger("httpware.retry").setLevel(logging.WARNING) logging.getLogger("httpware.bulkhead").setLevel(logging.WARNING) logging.getLogger("httpware.circuit_breaker").setLevel(logging.INFO) # INFO for recovery events logging.getLogger("httpware.timeout").setLevel(logging.WARNING) ``` -For OTel attribute enrichment on the active span — install the extra: +To also get the events on the active OpenTelemetry span, install the extra: ```bash pip install httpware[otel] ``` -When installed, `_emit_event` calls `trace.get_current_span().add_event(name, attributes=...)` automatically. We never create our own spans, so events only appear if something else creates one — see below for the minimal SDK + instrumentor setup that makes that happen. +httpware then adds each event to the current span with `trace.get_current_span().add_event(...)`. It never creates spans itself, so the events only show up when something else has started one. The next section shows the minimal setup that does. ## Wiring OpenTelemetry -`httpware[otel]` only ships `opentelemetry-api`. To make the observability events emitted by `AsyncRetry` and `AsyncBulkhead` visible, you also need: - -- An **SDK** (`opentelemetry-sdk`) to actually collect spans -- An **HTTP instrumentor** (`opentelemetry-instrumentation-httpx`) so each HTTP call creates a span — `httpware`'s events attach to that span via `trace.get_current_span().add_event(...)` +`httpware[otel]` only installs `opentelemetry-api`. To see the events you also need `opentelemetry-sdk` to collect spans, and `opentelemetry-instrumentation-httpx` to create a span for each HTTP call. httpware's events attach to that span. -Minimal setup (console exporter for development): +A minimal setup with a console exporter, for development: ```python from opentelemetry import trace @@ -53,6 +48,6 @@ trace.get_tracer_provider().add_span_processor(BatchSpanProcessor(ConsoleSpanExp HTTPXClientInstrumentor().instrument() ``` -After this runs, every `httpware` HTTP call gets an `HTTP ` span from the instrumentor, and AsyncRetry/AsyncBulkhead observability events appear as span events on it (no extra configuration needed in `httpware` itself — the events fire whenever an active span is present). +With this in place, the instrumentor gives every httpware call an `HTTP ` span, and the resilience middleware's events appear on it. httpware itself needs no configuration. -For production, swap `ConsoleSpanExporter` for your OTLP/Jaeger/Zipkin exporter. See the [OpenTelemetry Python docs](https://opentelemetry.io/docs/languages/python/) for the full SDK setup. +In production, replace `ConsoleSpanExporter` with your OTLP, Jaeger or Zipkin exporter. The [OpenTelemetry Python docs](https://opentelemetry.io/docs/languages/python/) cover the full SDK setup. diff --git a/docs/recipes/link-header-pagination.md b/docs/recipes/link-header-pagination.md index 56dbce5..d901411 100644 --- a/docs/recipes/link-header-pagination.md +++ b/docs/recipes/link-header-pagination.md @@ -1,8 +1,8 @@ # Link header pagination -GitLab, GitHub, and other APIs paginate via the [RFC 5988](https://datatracker.ietf.org/doc/html/rfc5988) `Link` response header: each page response carries a `Link: <…>; rel="next"` header pointing to the next page. To walk all pages you need both the decoded body **and** the response headers from the same call — `client.get(..., response_model=...)` returns only the body. +GitLab, GitHub and other APIs paginate with the [RFC 5988](https://datatracker.ietf.org/doc/html/rfc5988) `Link` header: each page carries a `Link: <…>; rel="next"` header that points to the next one. Walking the pages takes both the decoded body and the headers of each response, but `client.get(..., response_model=...)` returns only the body. -`send_with_response` returns both atomically. It routes the decoded body through the configured `ResponseDecoder`, so decoder failures surface as `DecodeError` — caught by `except httpware.ClientError` like every other failure mode. +`send_with_response` returns the response and the decoded body together. Decoding goes through the client's decoders as usual, so a bad body raises `DecodeError`, which `except httpware.ClientError` catches like any other failure. ## The pagination loop @@ -28,11 +28,11 @@ async def main() -> None: params = None # next link carries query ``` -`process` and `next_link` are caller-defined. Pick a Link-header parser that fits your project — there are several on PyPI, and the format is small enough to hand-roll. +`process` and `next_link` are yours to write. There are several Link header parsers on PyPI, and the format is small enough to parse by hand. ## Shorthand: per-verb `*_with_response` -When you do not need a pre-built `Request` object, the per-verb siblings collapse the `build_request` + `send_with_response` two-step into a single call: +If you don't need to build the `Request` yourself, each verb has a `*_with_response` method that does both steps in one call: ```python # two-step (pre-built request, required when you need full Request control) @@ -43,13 +43,13 @@ response, tags = await client.send_with_response(request, response_model=list[Ta response, tags = await client.get_with_response(url, params=params, response_model=list[Tag]) ``` -The full set of siblings is `get_with_response`, `post_with_response`, `put_with_response`, `patch_with_response`, `delete_with_response`, and `request_with_response`. There is no `head_with_response` or `options_with_response` — use `request_with_response` for those methods. +The methods are `get_with_response`, `post_with_response`, `put_with_response`, `patch_with_response`, `delete_with_response` and `request_with_response`. For HEAD and OPTIONS, use `request_with_response`. ## When to use which API -- **Body only, high-level verb:** `client.get(..., response_model=...)` -- **Body only, custom `Request`:** `client.send(request, response_model=...)` -- **Body + response metadata, simple URL:** `client.get_with_response(url, response_model=...)` -- **Body + response metadata, pre-built `Request`:** `client.send_with_response(request, response_model=...)` +| You need | From a URL | From a `Request` you built | +|---|---|---| +| The body | `client.get(url, response_model=...)` | `client.send(request, response_model=...)` | +| The body and the response | `client.get_with_response(url, response_model=...)` | `client.send_with_response(request, response_model=...)` | -`send_with_response` and the `*_with_response` siblings are not for streaming responses — use [`stream()`](../index.md#streaming-responses) for those. +None of these stream. For streaming responses, use [`stream()`](../index.md#streaming-responses). diff --git a/docs/recipes/modern-di.md b/docs/recipes/modern-di.md index fb2d9b9..0ef3d94 100644 --- a/docs/recipes/modern-di.md +++ b/docs/recipes/modern-di.md @@ -1,6 +1,6 @@ # Wiring `AsyncClient` into `modern-di` -If you wire your app's dependencies with [`modern-di`](https://modern-di.modern-python.org/) and want connection-pool teardown and middleware composition to flow through the container's lifecycle, this is the bridge. Both libraries ship under the [`modern-python`](https://github.com/modern-python) org. +This recipe registers httpware clients in a [`modern-di`](https://modern-di.modern-python.org/) container, so the container builds each client with its middleware and closes its connection pool on shutdown. Both libraries are part of the [`modern-python`](https://github.com/modern-python) org. ## The minimal wire-up @@ -29,25 +29,23 @@ async def main() -> None: await container.close_async() # runs the AsyncClient.aclose finalizer ``` -> **modern-di 2.x.** Resolution is sync — `container.resolve(...)`, no `await`. -> The root container is created plainly and torn down with `await -> container.close_async()` (the `async with` form is for -> `build_child_container(...)`, not the root). On modern-di 1.x, resolution was -> awaited; pin accordingly if you are still on 1.x. +!!! note "modern-di 2.x" + Resolution is sync: `container.resolve(...)`, with no `await`. Create the root + container directly and close it with `await container.close_async()`; the + `async with` form is for `build_child_container(...)`. In modern-di 1.x, + resolution was awaited. -Breaking that down: +- `Scope.APP` ties the client to the application's lifetime: one client per process, reusing its connection pool for every call. +- `cache_settings=providers.CacheSettings(...)` makes the provider a singleton. Without it, `Factory` builds a new `AsyncClient` on every resolve. +- `finalizer=AsyncClient.aclose` is the unbound async method. `modern-di` sees that it is async and awaits it when the container closes. -- **`Scope.APP`** ties the client to the application lifetime. One client per process; the connection pool is reused across all calls. -- **`cache_settings=providers.CacheSettings(...)`** is what makes the provider a singleton. Without it, `Factory` returns a fresh `AsyncClient` on every resolve. -- **`finalizer=AsyncClient.aclose`** is the unbound async method. `modern-di` detects the async finalizer and `await`s it on container teardown (here, on `close_async()`). +Don't write `finalizer=lambda c: c.aclose()`. The lambda is sync, so `modern-di` calls it without awaiting and the returned coroutine is dropped, leaking the connection pool. Pass the unbound method, or an `async def`. -A common first instinct here is `finalizer=lambda c: c.aclose()`. **That does not work** — the lambda itself is sync, so `modern-di` calls it synchronously and discards the returned coroutine unawaited. The underlying connection pool leaks. Pass the unbound async method directly, or wrap in `async def`. +The [`modern-di` factories docs](https://modern-di.modern-python.org/providers/factories/) cover `CacheSettings` in full, including scopes, `clear_cache`, and sync and async finalizers. -See the [`modern-di` factories docs](https://modern-di.modern-python.org/providers/factories/) for the broader `CacheSettings` story (scopes, `clear_cache`, sync vs async finalizers). +## A second backend collides on type -## Adding a second backend hits a type collision - -The obvious move when you talk to a second backend — register another `Factory(creator=AsyncClient, ...)` — fails at container construction: +Registering a second `Factory(creator=AsyncClient, ...)` for another backend fails when the container is built: ```python class ServiceClients(Group): @@ -70,11 +68,11 @@ class ServiceClients(Group): # . To resolve this issue: ... ``` -`modern-di` resolves dependencies by `bound_type`, which defaults to the creator's return type. Both providers default to `bound_type=AsyncClient` and collide in the providers registry. +`modern-di` resolves dependencies by `bound_type`, which defaults to the creator's return type, so both providers register as `AsyncClient`. -## Fix: one wrapper subclass per backend +## Fix: one subclass per backend -Give each provider a distinct `bound_type` by subclassing `AsyncClient`: +Subclass `AsyncClient` once per backend so each provider has its own `bound_type`: ```python from modern_di import Container, Group, Scope, providers @@ -115,15 +113,11 @@ async def main() -> None: await container.close_async() ``` -A couple of notes: - -- Subclasses are **typing-only**. Empty body, no overrides. They inherit `__init__`, `aclose`, and every HTTP method unchanged. -- Each `Factory` now has a distinct `bound_type`, so `container.resolve(UserApi)` and `container.resolve(BillingApi)` route to the right provider. -- `modern-di`'s error suggestions are subclass-aware. If a caller asks for `container.resolve(AsyncClient)` after only the subclasses are registered, the error message points them at the right subclass. +The subclasses are empty and exist only as types; they inherit everything from `AsyncClient`. `container.resolve(UserApi)` and `container.resolve(BillingApi)` now reach the right provider. If code asks for `container.resolve(AsyncClient)` when only the subclasses are registered, `modern-di`'s error message suggests the subclasses. ## Middleware in `kwargs=` -`AsyncClient`'s middleware chain is composed once at construction and frozen for the client's lifetime. With a singleton-scoped `Factory`, "once at construction" means "once per container build." Drop the middleware list into `kwargs=`: +A client's middleware is fixed when it is built, and a singleton `Factory` builds it once per container. Pass the middleware list in `kwargs=`: ```python from httpware import AsyncClient, AsyncBulkhead, AsyncRetry @@ -141,11 +135,10 @@ class ServiceClients(Group): ) ``` -Each cached singleton owns its own `AsyncBulkhead` and `AsyncRetry` state — what you want when different backends have different reliability profiles. +Each client gets its own `AsyncBulkhead` and `AsyncRetry`, so one backend's failures don't use up another's slots or retry budget. ## See also -- **[Quick-Start](../index.md)** — the base `AsyncClient` API. -- **[Middleware guide](../middleware.md)** — what `AsyncBulkhead` and `AsyncRetry` are doing in `kwargs[middleware]`. -- **[Resilience reference](../resilience.md)** — every parameter on `AsyncRetry`, `RetryBudget`, `AsyncBulkhead`. -- **[`modern-di` factories](https://modern-di.modern-python.org/providers/factories/)** — `CacheSettings`, scopes, the broader provider story. +- [Quickstart](../index.md): the `AsyncClient` API. +- [Resilience](../resilience.md): every resilience middleware and its parameters. +- [`modern-di` factories](https://modern-di.modern-python.org/providers/factories/): `CacheSettings`, scopes and other provider options. diff --git a/docs/recipes/phase-decorator-patterns.md b/docs/recipes/phase-decorator-patterns.md index 468900a..6e23a47 100644 --- a/docs/recipes/phase-decorator-patterns.md +++ b/docs/recipes/phase-decorator-patterns.md @@ -1,14 +1,12 @@ # Phase decorator recipes -The `@async_before_request`, `@async_after_response`, and `@async_on_error` decorators from `httpware.middleware` turn a single async function into an `AsyncMiddleware`. Reach for them when the logic fits one function — no `self` state, no need to bracket `await next(...)` from both sides. +`@async_before_request`, `@async_after_response` and `@async_on_error` turn a single async function into an `AsyncMiddleware`. Use them when the logic fits in one function, with no state on `self` and no code on both sides of `await next(...)`. -When the same logic would fit `httpx2.event_hooks` instead, prefer the hook: it's a layer below httpware's exception mapping and chain ordering, and is the right place for transforms that don't need either. Phase decorators participate in the middleware chain — they see `httpware` exceptions (mapped from `httpx2` ones), and they compose with `AsyncRetry`, `AsyncBulkhead`, and other middleware in a documented order. - -This page collects four worked recipes — one minimal and one realistic for `@async_before_request`, a response-status counter for `@async_after_response`, and a `NetworkError` fallback for `@async_on_error`. +Phase decorators run in the middleware chain: they see httpware exceptions and take their place in the chain order next to `AsyncRetry` and `AsyncBulkhead`. Logic that needs neither can also be an `httpx2` event hook; [Middleware](../middleware.md#choosing-where-behavior-lives) explains the difference. ## `@async_before_request`: bearer token -The smallest useful case — add a static `Authorization` header to every outgoing request. +Add a fixed `Authorization` header to every request: ```python import httpx2 @@ -31,11 +29,11 @@ async def main() -> None: await client.get("/me") ``` -`add_bearer` is now an `AsyncMiddleware` instance; pass it directly into `middleware=[…]`. Order in the list is outer→inner — if you add `AsyncRetry()` after, the bearer header is set on every retry attempt (which is what you want — each attempt is a real HTTP call and needs auth). +`add_bearer` is now an `AsyncMiddleware`, so it goes straight into `middleware=[...]`. If you add `AsyncRetry()` after it, every retry attempt carries the header too. ## `@async_before_request`: correlation ID from `contextvars` -A more realistic case — propagate a correlation ID set by your application's surrounding context (FastAPI middleware, structlog binder, etc.). The decorator pulls the ID out of a `ContextVar` and stamps it on the outgoing request. +Your application may already keep a correlation ID in a `ContextVar`, set by FastAPI middleware or a structlog binder. This middleware copies it onto each outgoing request: ```python import contextvars @@ -69,14 +67,13 @@ async def main() -> None: await client.get("/me") # request carries X-Correlation-Id: abc-123 ``` -Two notes worth calling out: +`propagate_correlation_id` runs once per call, before `AsyncRetry`, and the request it changes is the one every attempt resends, so all attempts share the header. -- **Placement matters.** `propagate_correlation_id` sits *before* `AsyncRetry` in the chain, so it re-runs for each retry attempt. The header is set on every attempt, but the ID itself stays the same across attempts because `ContextVar` state doesn't change between them. -- **vs `event_hooks`.** This is also expressible as `event_hooks={"request": [propagate_correlation_id]}` on the wrapped httpx2 client, with one functional difference: hooks run *below* the httpware chain, so they fire on every transport attempt including post-redirect hops. For correlation IDs the behaviour is usually equivalent; for anything that should fire once per *logical* call (e.g. a UUID generated inline), the phase decorator is correct. +The undecorated function would also work as a request hook, `event_hooks={"request": [...]}`, on an `httpx2` client you pass as `httpx2_client=`. Hooks run below the chain, on every transport attempt and every redirect hop. That makes no difference for a correlation ID read from a `ContextVar`, but a value generated in the function, like a fresh UUID, would change on each attempt. ## `@async_after_response`: counter by status class -Side-effect-only recipe — increment a counter keyed by status class (`2xx`, `4xx`, `5xx`) every time a response comes back. The decorator returns the response unchanged. +Count responses by status class (`2xx`, `4xx`, `5xx`) and return each response unchanged: ```python from collections.abc import Callable @@ -112,16 +109,14 @@ async def main() -> None: await client.get("/me") ``` -Notes: - -- The factory function `status_class_counter(metric_sink)` is the canonical way to parameterize a phase decorator — the decorated function itself takes no extra args, but the enclosing factory can. -- **Why not request *latency* here?** Wall-clock timing requires bracketing the `await next(request)` call from both sides — `@async_after_response` only sees the response on the way back, so it can't measure the call duration. (`response.elapsed` from httpx2 is also unavailable at this chain point because the body isn't read yet.) Use a raw `AsyncMiddleware` class for timing — see [middleware.md](../middleware.md) for that pattern. -- Wiring real sinks: pass `statsd.incr` (statsd), a `prometheus_client.Counter` `.inc` method (Prometheus), or `datadog.statsd.increment` (Datadog). The signature `Callable[[str, int], None]` is loose on purpose. -- This middleware does NOT see exceptions. Failed requests (caught by `AsyncRetry`, raised as `StatusError`, or surfaced as `NetworkError`) never reach `@async_after_response`. If you want counts of *attempted* requests including failures, install a `httpware.retry` log handler or write a raw `AsyncMiddleware` that brackets the call. +- The decorated function can't take extra arguments, so a factory such as `status_class_counter(metric_sink)` is how you pass it settings. +- It can't measure latency. Timing needs code before and after `await next(request)`, and `@async_after_response` only runs after. `response.elapsed` isn't set yet either, because the body hasn't been read at this point. Write an `AsyncMiddleware` class for timing; see [Middleware](../middleware.md). +- `metric_sink` can be `statsd.incr`, the `.inc` method of a `prometheus_client.Counter`, or `datadog.statsd.increment`; the `Callable[[str, int], None]` type is kept loose for that reason. +- It never sees failures. A request that ends in a `StatusError` or `NetworkError` never reaches `@async_after_response`. To count failed requests too, write an `AsyncMiddleware` class that wraps the call, or add a handler on the `httpware.retry` logger. ## `@async_on_error`: fallback on `NetworkError` -When the upstream is unreachable, return a synthesized 503 with a sentinel header so callers can branch on degraded mode. The decorator returns a `Response` on `NetworkError`, and `None` for everything else (re-raise). +When the upstream is unreachable, return a made-up 503 with a marker header so callers can switch to a degraded mode. The function returns a `Response` for `NetworkError` and `None`, which re-raises, for anything else. ```python import httpx2 @@ -156,15 +151,13 @@ async def main() -> None: ... # degraded path ``` -Notes: - -- **`return None` re-raises.** The decorator only synthesizes a response for cases it actually handles; everything else propagates unchanged. Be specific about which exception types you absorb. -- **Returning a 4xx/5xx response does NOT re-trigger status mapping.** The terminal raises `StatusError` on the upstream response; once your `@async_on_error` returns, the synthesized response flows up the chain unchanged. If you want callers to see a `ServiceUnavailableError`, raise it directly instead of synthesizing. -- **Catches `Exception`, not `BaseException`.** `asyncio.CancelledError` propagates — your fallback won't accidentally swallow cooperative cancellation. -- **Placement vs `AsyncRetry`.** Put `@async_on_error` *outside* `AsyncRetry` (`middleware=[fallback_on_network_error, AsyncRetry()]`) if you want the fallback to apply only after all retries have failed. Inside `AsyncRetry` (`middleware=[AsyncRetry(), fallback_on_network_error]`) the fallback fires on the first network error and `AsyncRetry` never sees it. The outer placement is almost always what you want. +- Returning `None` re-raises the exception. Handle only the exception types you mean to absorb. +- A 4xx or 5xx response you return is not turned into a `StatusError`. Status mapping happens once, on the upstream response, and your response travels up the chain as is. If callers should get a `ServiceUnavailableError`, raise it. +- The decorator catches `Exception`, not `BaseException`, so `asyncio.CancelledError` still propagates. +- Put the fallback before `AsyncRetry`, as in `middleware=[fallback_on_network_error, AsyncRetry()]`, so it only runs once all retries have failed. After `AsyncRetry`, it handles the first network error and `AsyncRetry` never gets to retry. ## See also -- **[Middleware guide](../middleware.md)** — the protocol contract, the raw-`AsyncMiddleware` class form, and "when NOT to write a middleware". -- **[Resilience reference](../resilience.md)** — `AsyncRetry`, `RetryBudget`, `AsyncBulkhead` parameters and behaviour. -- **[Errors guide](../errors.md)** — `NetworkError`, `StatusError`, and the full exception tree. +- [Middleware](../middleware.md): the middleware protocol, writing a middleware class, and when to use something else. +- [Resilience](../resilience.md): the built-in resilience middleware and their parameters. +- [Errors](../errors.md): `NetworkError`, `StatusError` and the rest of the exception tree. diff --git a/docs/resilience.md b/docs/resilience.md index 1ade8c4..5adec2d 100644 --- a/docs/resilience.md +++ b/docs/resilience.md @@ -1,24 +1,19 @@ # Resilience reference -`httpware` ships these resilience primitives under `httpware.middleware.resilience`, all composable through the standard [Middleware](middleware.md) / [AsyncMiddleware](middleware.md) chain: +httpware's resilience middleware live in `httpware.middleware.resilience` and plug into the same [middleware](middleware.md) chain as your own: -- **`Retry` / `AsyncRetry`** — automatic retry of transient failures with full-jitter exponential backoff -- **`RetryBudget`** — Finagle-style token bucket bounding the global retry rate to prevent retry storms; safe to share across sync `Client` and `AsyncClient` in the same process -- **`Bulkhead` / `AsyncBulkhead`** — concurrency limiter with bounded acquire-wait (`threading.Semaphore` and `asyncio.Semaphore` respectively) +- [`AsyncRetry`](#asyncretry) and [`Retry`](#retry) retry transient failures with full-jitter exponential backoff. +- [`RetryBudget`](#retrybudget) is a token bucket that caps the share of traffic spent on retries, so retries can't pile onto an outage. One instance can be shared by sync and async clients. +- [`AsyncBulkhead`](#asyncbulkhead) and [`Bulkhead`](#bulkhead) cap concurrent requests and reject a request that can't get a slot in time. +- [`AsyncCircuitBreaker` and `CircuitBreaker`](#asynccircuitbreaker-circuitbreaker) stop sending requests to a downstream that keeps failing. +- [`AsyncKeyedCircuitBreaker` and `KeyedCircuitBreaker`](#asynckeyedcircuitbreaker-keyedcircuitbreaker) keep a separate circuit per upstream. +- [`AsyncTimeout`](#asynctimeout) bounds the total time of a call, retries included. -A key ordering constraint: `AsyncBulkhead` must sit outside `AsyncRetry` (before it in `middleware=`) so one slot covers all retry attempts of a single call. For the full recommended ordering across all four primitives, see [Composition](#composition). Reach for the [Middleware guide](middleware.md) when you want to write your own resilience policy. +Order matters. [Composition](#composition) gives the recommended order and the reason for each position. !!! tip "See it under load" - New to these patterns? The [interactive demos](demos/index.md) show each one - surviving an outage side by side with an unprotected client. - -- [`AsyncRetry`](#asyncretry) -- [`RetryBudget`](#retrybudget) -- [`AsyncBulkhead`](#asyncbulkhead) -- [`AsyncCircuitBreaker` / `CircuitBreaker`](#asynccircuitbreaker-circuitbreaker) -- [`AsyncKeyedCircuitBreaker` / `KeyedCircuitBreaker`](#asynckeyedcircuitbreaker-keyedcircuitbreaker) -- [`AsyncTimeout`](#asynctimeout) -- [Sync `Retry` and `Bulkhead`](#sync-retry-and-bulkhead) + The [interactive demos](demos/index.md) run each pattern through an outage + next to a client without it. ## `AsyncRetry` @@ -28,39 +23,40 @@ from httpware.middleware.resilience import AsyncRetry | Parameter | Default | Effect | |---|---|---| -| `max_attempts` | `3` | Total tries (including the first). `1` disables retries entirely; `<1` raises `ValueError`. | +| `max_attempts` | `3` | Total tries, including the first. `1` disables retries; `<1` raises `ValueError`. | | `base_delay` | `0.1` (s) | Floor for the full-jitter exponential backoff. | | `max_delay` | `5.0` (s) | Ceiling for backoff. | | `retry_status_codes` | `frozenset({408, 429, 502, 503, 504})` | Status codes considered retryable. | -| `retry_methods` | `frozenset({"GET", "HEAD", "OPTIONS", "PUT", "DELETE"})` | Idempotent methods only by default. POST excluded; pass an explicit frozenset including `"POST"` to retry it. | -| `respect_retry_after` | `True` | When a retryable response carries a `Retry-After` header, sleep for that value instead of the jittered backoff. If it exceeds `max_delay`, AsyncRetry gives up and re-raises the underlying `StatusError`, attaching an exception note (PEP 678): `httpware: Retry-After (Ns) exceeded max_delay (Ms); giving up`. Opt out with `respect_retry_after=False` or a higher `max_delay`. | -| `budget` | `RetryBudget()` (default-configured) | The token bucket. Pass a shared `RetryBudget` instance to apply one budget across multiple clients. | +| `retry_methods` | `frozenset({"GET", "HEAD", "OPTIONS", "PUT", "DELETE"})` | Idempotent methods only. To retry POST, pass a frozenset that includes `"POST"`. | +| `respect_retry_after` | `True` | When a retryable response has a `Retry-After` header, wait that long instead of the jittered backoff. If the value is over `max_delay`, `AsyncRetry` gives up and re-raises the `StatusError` with a PEP 678 note: `httpware: Retry-After (Ns) exceeded max_delay (Ms); giving up`. To avoid this, set `respect_retry_after=False` or raise `max_delay`. | +| `budget` | `RetryBudget()` | The token bucket. Pass one shared `RetryBudget` to several clients to give them a joint budget. | -For a whole-operation wall-clock bound across all retry attempts, compose `AsyncTimeout` outermost — see [AsyncTimeout](#asynctimeout) below. For a per-request bound, use `httpx2.Timeout` on the client or pass `timeout=` per request. +To bound the total time across all attempts, put [`AsyncTimeout`](#asynctimeout) first in the chain. To bound a single request, set `httpx2.Timeout` on the client or pass `timeout=` per request. ### Retry-After parsing -`Retry-After` is parsed as either: -- **Integer seconds** — `Retry-After: 30` → sleep 30s -- **HTTP-date** (RFC 5322) — `Retry-After: Wed, 21 Oct 2026 07:28:00 GMT` → sleep until that absolute time, computed delay floored at 0 +`Retry-After` can be either: + +- integer seconds: `Retry-After: 30` waits 30 seconds. +- an HTTP date (RFC 5322): `Retry-After: Wed, 21 Oct 2026 07:28:00 GMT` waits until that time, or not at all if it has passed. -Either form triggers the same give-up-and-re-raise rule above if it exceeds `max_delay`. Negative integer values floor at 0; malformed values are ignored, falling back to the jittered backoff. +Either form over `max_delay` makes `AsyncRetry` give up as described above. Negative integers count as 0. Malformed values are ignored and the jittered backoff is used instead. ### Streaming-body refusal -If the request body was an async-iterable, `AsyncRetry` refuses to retry — the iterator is consumed after the first attempt and can't replay. The original exception is re-raised with a PEP 678 note: +If the request body was an async iterable, `AsyncRetry` does not retry, because the first attempt consumed the iterator. It re-raises the original exception with a PEP 678 note: ``` -httpware: not retrying — request body is a stream that cannot replay across attempts +httpware: not retrying because the request body is a stream that cannot replay across attempts ``` -A non-idempotent request that also carries a streaming body is refused first by the method-eligibility check — that early exit re-raises the original exception without the streaming-refusal note. The note (and the `httpware.retry` `retry.streaming_refused` observability event) is added only on the retryable-failure path, i.e. once the method and status are both eligible — see [Observability](observability.md). +The note, and the `retry.streaming_refused` event on `httpware.retry`, appear only when the failure would otherwise have been retried. A POST with a streaming body, for example, is re-raised without the note because POST is not retried in the first place. See [Observability](observability.md). ### Exhaustion behavior -On exhaustion, `AsyncRetry` re-raises the *last* exception observed (e.g., `ServiceUnavailableError`, `NetworkError`), preserving the original class so `except ServiceUnavailableError` still catches it. A PEP 678 note is added: `httpware: gave up after N attempts`. +When attempts run out, `AsyncRetry` re-raises the last exception, such as `ServiceUnavailableError` or `NetworkError`, with its original class, so `except ServiceUnavailableError` still works. It adds the note `httpware: gave up after N attempts`. -If exhaustion is caused by the budget refusing a retry (not by `max_attempts`), the raised exception is `RetryBudgetExhaustedError` instead, with `last_response` / `last_exception` / `attempts` fields populated. See the [Errors reference](errors.md). +If the budget refuses a retry before `max_attempts` is reached, `AsyncRetry` raises `RetryBudgetExhaustedError` instead, with `last_response`, `last_exception` and `attempts` set. See [Errors](errors.md). ## `RetryBudget` @@ -68,12 +64,12 @@ If exhaustion is caused by the budget refusing a retry (not by `max_attempts`), from httpware.middleware.resilience import RetryBudget ``` -A Finagle-style token bucket bounding retry rate. Each request deposits a token; each retry attempts to withdraw one. Available retries are bounded by `percent_can_retry` of recent deposits, plus a `min_retries_per_sec * ttl` floor. +A token bucket in the style of Finagle's retry budget. Each request deposits a token and each retry tries to withdraw one. Retries are allowed up to `percent_can_retry` of recent deposits, plus a floor of `min_retries_per_sec * ttl`. | Parameter | Default | Effect | |---|---|---| | `ttl` | `10.0` (s) | Sliding window over which deposits and withdrawals count. | -| `min_retries_per_sec` | `10.0` | Absolute floor — at least this many retries/sec are permitted regardless of deposit rate. | +| `min_retries_per_sec` | `10.0` | Floor: at least this many retries per second are allowed, whatever the deposit rate. | | `percent_can_retry` | `0.2` | Fraction of recent deposits that can convert to retries (above the floor). | ### The token-bucket formula @@ -82,15 +78,15 @@ A Finagle-style token bucket bounding retry rate. Each request deposits a token; ceiling = ceil(len(deposits_in_window) * percent_can_retry) + int(min_retries_per_sec * ttl) ``` -The percent term rounds **up** (`math.ceil`); the floor term truncates (`int`). A withdrawal fails when `len(withdrawn_in_window) >= ceiling`. +The percent term rounds up (`math.ceil`); the floor term truncates (`int`). A withdrawal fails when `len(withdrawn_in_window) >= ceiling`. ### Why a floor matters -If the deposit rate is zero (no traffic yet), the percent term is zero — without the floor, the very first retry would be refused. The floor lets small-traffic clients still retry on isolated failures; high-traffic clients are dominated by the percent term and the floor becomes irrelevant. +With no traffic yet, the percent term is zero, and without the floor the first retry would be refused. The floor lets low-traffic clients retry occasional failures. For high-traffic clients the percent term is much larger and the floor stops mattering. ### Sharing across clients -Pass the same `RetryBudget` instance to multiple `AsyncClient`s when they hit the same downstream — one joint budget covers them all: +Clients that call the same downstream can share one `RetryBudget`: ```python import asyncio @@ -112,7 +108,7 @@ async def main() -> None: ### Thread safety -`RetryBudget` is thread-safe and asyncio-safe — all mutations go through a `threading.Lock`. A single instance is safe to share across threads, across coroutines on one event loop, and across `Client` / `AsyncClient` pairs in the same process. See [Sync Retry and Bulkhead](#sync-retry-and-bulkhead) for the cross-world sharing pattern. +`RetryBudget` guards its state with a `threading.Lock`, so one instance can be shared across threads, across coroutines, and between a `Client` and an `AsyncClient` in the same process. ## `AsyncBulkhead` @@ -120,20 +116,20 @@ async def main() -> None: from httpware.middleware.resilience import AsyncBulkhead ``` -Concurrency limiter via `asyncio.Semaphore`. Acquires a slot before each request (bounded by `acquire_timeout`); releases on success, exception, AND cancellation. +Limits concurrent requests with an `asyncio.Semaphore`. Each request waits up to `acquire_timeout` for a slot and releases it when it finishes, fails or is cancelled. | Parameter | Default | Effect | |---|---|---| -| `max_concurrent` | **REQUIRED** | Maximum in-flight requests. `<1` raises `ValueError`. No default — the right cap depends on downstream capacity and SLA. | +| `max_concurrent` | required | Maximum requests in flight. `<1` raises `ValueError`. There is no default because the right cap depends on the downstream's capacity. | | `acquire_timeout` | `1.0` (s) | How long to wait for a slot before raising `BulkheadFullError`. `None` waits forever; `0` fails fast. `<0` raises `ValueError`. | ### Slot release contract -The slot is released in a `try/finally` around `await next(request)`, so success, an exception propagating, or a `CancelledError` propagating all release it deterministically — it cannot leak. +The slot is released in a `try/finally` around `await next(request)`, so a success, an exception or a `CancelledError` all release it. ### Sharing across clients -Same pattern as `RetryBudget`. One instance, many clients: +Like `RetryBudget`, one bulkhead can be shared by several clients: ```python shared_bulkhead = AsyncBulkhead(max_concurrent=10) @@ -147,7 +143,7 @@ async with ( ### Rejection -When `acquire_timeout` elapses without a slot opening, `AsyncBulkhead` raises `BulkheadFullError` (carries the configured `max_concurrent` and `acquire_timeout` for caller logging). See the [Errors reference](errors.md). The `httpware.bulkhead` `bulkhead.rejected` observability event fires at the same site — see [Observability](observability.md). +If no slot opens within `acquire_timeout`, `AsyncBulkhead` raises `BulkheadFullError` with the configured `max_concurrent` and `acquire_timeout` (see [Errors](errors.md)) and emits `bulkhead.rejected` on `httpware.bulkhead` (see [Observability](observability.md)). ## `AsyncCircuitBreaker` / `CircuitBreaker` @@ -156,13 +152,13 @@ from httpware.middleware.resilience import AsyncCircuitBreaker # async from httpware.middleware.resilience import CircuitBreaker # sync ``` -Classic consecutive-failure circuit breaker. Counts failures and prevents requests from reaching a downstream that is known to be broken. +A circuit breaker counts consecutive failures and, past a threshold, stops sending requests to the downstream for a while. ### States -- **CLOSED** — normal operation. Each counted failure increments the consecutive-failure counter. Once `failure_threshold` consecutive counted failures accumulate, the circuit opens. -- **OPEN** — fast-fail. While elapsed time is below `reset_timeout`, requests are rejected immediately with `CircuitOpenError` (carrying `retry_after` seconds until the next probe window). The first request after `reset_timeout` elapses transitions the circuit to HALF_OPEN and becomes the probe. -- **HALF_OPEN** — exactly one probe is admitted. If `success_threshold` consecutive probe successes are observed, the circuit closes. A single probe failure re-opens the circuit. +- CLOSED: normal operation. After `failure_threshold` counted failures in a row, the circuit opens. +- OPEN: until `reset_timeout` has passed, every request fails at once with `CircuitOpenError`, whose `retry_after` says how many seconds remain. The first request after that moves the circuit to HALF_OPEN and becomes the probe. +- HALF_OPEN: one probe at a time is let through. After `success_threshold` successful probes in a row the circuit closes; one failed probe opens it again. ### Constructor @@ -171,20 +167,20 @@ Classic consecutive-failure circuit breaker. Counts failures and prevents reques | `failure_threshold` | `5` | Consecutive counted failures required to open. `<1` raises `ValueError`. | | `reset_timeout` | `30.0` (s) | Seconds to stay OPEN before admitting a probe. `<0` raises `ValueError`. | | `success_threshold` | `1` | Consecutive probe successes required to close. `<1` raises `ValueError`. | -| `failure_status_codes` | `None` | Which status codes count as failures. `None` → all 5xx (`500`–`599`). | -| `failure_rate_threshold` | `None` | Opts into time-based rate mode when set (see [below](#time-based-failure-rate-mode)). Fraction of failures in the rolling window that opens the circuit; `None` keeps classic consecutive-failure mode. | -| `window_seconds` | `30.0` (s) | Rate mode only: width of the rolling window `failure_rate_threshold` is measured over. | -| `minimum_calls` | `20` | Rate mode only: outcomes required in the window before the rate is evaluated. | +| `failure_status_codes` | `None` | Status codes that count as failures. `None` means every 5xx. | +| `failure_rate_threshold` | `None` | Setting it switches to [rate mode](#time-based-failure-rate-mode): the share of failures in the rolling window that opens the circuit. `None` keeps the consecutive-failure mode. | +| `window_seconds` | `30.0` (s) | Rate mode only: the length of the rolling window. | +| `minimum_calls` | `20` | Rate mode only: how many outcomes the window needs before the rate is checked. | ### Failure classification -A **counted failure** is a `NetworkError`, an httpware `TimeoutError`, or a `StatusError` whose status code is in `failure_status_codes`. All other exceptions propagate without affecting circuit state. +A counted failure is a `NetworkError`, an httpware `TimeoutError`, or a `StatusError` whose status is in `failure_status_codes`. Other exceptions pass through without changing the circuit. -**4xx responses — including 429 — count as successes.** A 429 means the service is healthy but throttling; tripping the circuit on it would amplify an incident by adding circuit-open rejections on top of the throttle. +4xx responses, 429 included, count as successes. A 429 means the service is up and throttling you; opening the circuit on it would add circuit-open rejections on top of the throttling. ### `CircuitOpenError` -Raised when the circuit is OPEN (with a positive `retry_after: float`) or when HALF_OPEN with a probe already in flight (`retry_after=None`). Inherits `httpware.ClientError`. See the [Errors reference](errors.md). +Raised while the circuit is OPEN, with a positive `retry_after`, or while it is HALF_OPEN and a probe is already in flight, with `retry_after=None`. It subclasses `httpware.ClientError`; see [Errors](errors.md). ### Observability @@ -192,16 +188,16 @@ Emitted on logger `httpware.circuit_breaker`: | Event | When | |---|---| -| `circuit.opened` | Failure threshold reached; circuit transitions CLOSED → OPEN | -| `circuit.rejected` | Request fast-failed (OPEN or HALF_OPEN probe slot taken) | -| `circuit.half_open` | Reset timeout elapsed; circuit transitions OPEN → HALF_OPEN | -| `circuit.closed` | Success threshold reached; circuit transitions HALF_OPEN → CLOSED | +| `circuit.opened` | The failure threshold was reached and the circuit went from CLOSED to OPEN. | +| `circuit.rejected` | A request failed fast because the circuit was OPEN or a HALF_OPEN probe was already running. | +| `circuit.half_open` | The reset timeout passed and the circuit went from OPEN to HALF_OPEN. | +| `circuit.closed` | The success threshold was reached and the circuit went from HALF_OPEN to CLOSED. | ### Time-based failure-rate mode -By default the circuit breaker trips on `failure_threshold` *consecutive* counted failures. This can miss partial degradation: a downstream returning errors on exactly half of all requests will never form a consecutive streak long enough to trip — the circuit stays closed while the error rate sits at 50%. +By default the breaker opens after `failure_threshold` failures in a row. A downstream that fails every other request never produces a long enough streak, so the circuit stays closed at a 50% error rate. -Passing `failure_rate_threshold` switches to rate mode (params in the [constructor table](#constructor) above): +Passing `failure_rate_threshold` switches to rate mode, which opens on the share of failures in a rolling window (parameters in the [constructor table](#constructor)): ```python from httpware.middleware.resilience import AsyncCircuitBreaker @@ -214,11 +210,11 @@ breaker = AsyncCircuitBreaker( ) ``` -Classic mode is the default; `failure_threshold` is ignored once rate mode is active. Half-open recovery works identically in both modes. The same `CircuitBreaker` constructor accepts the same parameters for sync clients. +In rate mode `failure_threshold` is ignored. Recovery through HALF_OPEN works the same in both modes, and the sync `CircuitBreaker` takes the same parameters. ### State introspection -Both `AsyncCircuitBreaker` and `CircuitBreaker` expose a read-only `state` property returning a public `CircuitState` enum: +`AsyncCircuitBreaker` and `CircuitBreaker` have a read-only `state` property that returns a `CircuitState`: ```python from httpware import CircuitState @@ -230,11 +226,11 @@ if breaker.state is CircuitState.OPEN: ... # report the dependency as degraded ``` -`state` reflects the stored state at the moment of the call and is read-only (writing raises `AttributeError`). The OPEN→HALF_OPEN transition is lazy — it fires only once a request is actually admitted after `reset_timeout` elapses, not on a clock tick — so `state` keeps reporting `OPEN` until that happens; reading it never triggers the transition. The same property exists on the sync `CircuitBreaker`. +Assigning to `state` raises `AttributeError`. The move from OPEN to HALF_OPEN happens when the first request arrives after `reset_timeout`, not when the timeout passes, so `state` reports `OPEN` until then. Reading `state` never changes it. ### Sharing -Pass the same instance to multiple clients to enforce one shared circuit across them. A `CircuitBreaker` (sync) cannot be shared with an `AsyncCircuitBreaker` — they use different concurrency primitives. +Pass the same instance to several clients to give them one circuit. A sync `CircuitBreaker` and an `AsyncCircuitBreaker` cannot share state, because they use different locking primitives. ### Example @@ -252,7 +248,7 @@ async with AsyncClient( response = await client.get("/users/1") ``` -Sync usage is identical: `Client` + `CircuitBreaker`, no `await`. +For sync code, use `Client` and `CircuitBreaker` without `await`. ## `AsyncKeyedCircuitBreaker` / `KeyedCircuitBreaker` @@ -295,7 +291,7 @@ async with AsyncClient( await client.get("https://suggest-b.example/v1/suggest") # unaffected if suggest-a is down ``` -In the [composition](#composition), the keyed breaker goes where the circuit breaker goes. It sits outside `AsyncRetry` and counts one outcome per retry sequence. Sync usage is identical: `Client` + `KeyedCircuitBreaker`, no `await`. +In the [composition order](#composition), the keyed breaker takes the circuit breaker's place, before `AsyncRetry`, so it counts one outcome per retry sequence. For sync code, use `Client` and `KeyedCircuitBreaker` without `await`. ## `AsyncTimeout` @@ -303,29 +299,29 @@ In the [composition](#composition), the keyed breaker goes where the circuit bre from httpware.middleware.resilience import AsyncTimeout ``` -Bounds total wall-clock time across the entire inner pipeline. Place it outermost to enforce "this whole operation must finish within `timeout` seconds, even across retries and backoff sleeps." On expiry it raises `httpware.TimeoutError`. +Bounds the total time of everything after it in the chain. Put it first so a call, with all its retries and backoff, finishes within `timeout` seconds. When time runs out it raises `httpware.TimeoutError`. | Parameter | Default | Effect | |---|---|---| -| `timeout` | **REQUIRED** | Overall deadline in seconds. Must be a finite number `> 0`; a non-finite (`inf`/`nan`) or `≤0` value raises `ValueError`. | +| `timeout` | required | Overall deadline in seconds. Must be finite and greater than 0, or `ValueError` is raised. | -**This is not a per-call timeout.** httpx2's connect/read/write/pool timeouts are the right tool for bounding a single outbound call; `AsyncTimeout` doesn't duplicate them. What httpx2 cannot bound is the total wall-clock across a whole retry sequence — `AsyncTimeout` fills that gap. +`AsyncTimeout` does not replace per-request timeouts. httpx2's connect, read, write and pool timeouts bound each request; `AsyncTimeout` bounds the whole retry sequence, which httpx2 can't. -**No sync `Timeout` exists.** Sync Python has no cancellation primitive that can interrupt a blocking httpx2 call mid-flight. For sync per-call bounds, configure `httpx2.Timeout` on the wrapped client or pass `timeout=` per request. +There is no sync `Timeout`, because sync Python cannot interrupt a blocking httpx2 call midway. In sync code, set `httpx2.Timeout` on the client or pass `timeout=` per request. -Observability event: `timeout.exceeded` on logger `httpware.timeout`. See [Composition](#composition) below for a worked example placing `AsyncTimeout` outermost alongside the other primitives. +On expiry it emits `timeout.exceeded` on `httpware.timeout`. [Composition](#composition) has an example with `AsyncTimeout` first in the chain. ## Composition -The recommended ordering (not enforced, but each position has a reason): +httpware doesn't enforce an order, but this one is recommended: ``` AsyncTimeout → AsyncCircuitBreaker → AsyncBulkhead → AsyncRetry → terminal ``` -- `AsyncTimeout` outermost so the overall deadline covers the entire sequence including retries and backoff. -- `AsyncCircuitBreaker` outside `AsyncRetry` so an open circuit short-circuits the whole retry loop without attempting any calls. This also means the breaker counts one outcome per fully-exhausted retry sequence rather than one per individual attempt. Placing it outside `AsyncBulkhead` too means a request the open circuit rejects never consumes a concurrency slot. -- `AsyncBulkhead` outside `AsyncRetry` so one slot covers all retry attempts of a single call. Flip those two (`[AsyncRetry, AsyncBulkhead]`) and each retry grabs a fresh slot — defeating the bulkhead under load. +- `AsyncTimeout` comes first, so the deadline covers every retry and backoff. +- `AsyncCircuitBreaker` comes before `AsyncRetry`, so an open circuit rejects the call without starting the retry loop, and the breaker counts one outcome per retry sequence instead of one per attempt. Putting it before `AsyncBulkhead` too means a rejected call never takes a slot. +- `AsyncBulkhead` comes before `AsyncRetry`, so one slot covers every attempt of a call. In the reverse order each retry takes a new slot, and under load the bulkhead admits more work than its cap intends. ```python from httpware import AsyncClient @@ -350,11 +346,11 @@ async def main() -> None: await client.get("/users/1") ``` -Cross-cutting middleware that emit per-call state (e.g., the Request-ID middleware in the [Middleware guide](middleware.md)) should sit outside `AsyncRetry` for the same reason — so all attempts of one call share one ID rather than getting a fresh ID per attempt. +Your own middleware that sets something per call, like the request-ID middleware in [Middleware](middleware.md), also belongs before `AsyncRetry`, so every attempt of a call shares one ID. ## Sync Retry and Bulkhead -The sync flavors mirror the async ones for use with `Client`. +Use these with the sync `Client`. ### `Retry` @@ -362,9 +358,9 @@ The sync flavors mirror the async ones for use with `Client`. from httpware.middleware.resilience import Retry ``` -`Retry` takes the identical parameters as `AsyncRetry` (table [above](#asyncretry)); it sleeps with `time.sleep` between attempts. `Retry-After`, streaming-body refusal, exhaustion behavior, and `RetryBudgetExhaustedError` semantics are identical to `AsyncRetry`. +`Retry` takes the same parameters as [`AsyncRetry`](#asyncretry) and sleeps with `time.sleep` between attempts. `Retry-After` handling, streaming bodies, giving up, and `RetryBudgetExhaustedError` all work the same way. -For a whole-attempt wall-clock bound, use `httpx2.Timeout` on the wrapped client or pass `timeout=` per request. No sync `Timeout` middleware exists — sync Python has no cancellation primitive that can interrupt a blocking call mid-flight. +There is no sync `Timeout` (see [`AsyncTimeout`](#asynctimeout)); set `httpx2.Timeout` on the client or pass `timeout=` per request. ### `Bulkhead` @@ -372,16 +368,16 @@ For a whole-attempt wall-clock bound, use `httpx2.Timeout` on the wrapped client from httpware.middleware.resilience import Bulkhead ``` -`Bulkhead` mirrors `AsyncBulkhead` (table [above](#asyncbulkhead)) on a `threading.Semaphore`. Slot release follows the same `try/finally` contract — success, exception, and (in sync land) interrupt-style exceptions all release the slot. +`Bulkhead` is [`AsyncBulkhead`](#asyncbulkhead) on a `threading.Semaphore`. The slot is released in the same `try/finally`, including on `KeyboardInterrupt` and other interrupts. -> **Per-world Bulkhead.** `Bulkhead` and `AsyncBulkhead` are separate primitives (`threading.Semaphore` vs `asyncio.Semaphore`); one instance cannot cap sync + async clients jointly. For a shared cap across both, create one of each with matching `max_concurrent` — the OS won't coordinate them, but the policy intent is documented. +One bulkhead cannot cap sync and async clients together, because `Bulkhead` and `AsyncBulkhead` use different semaphores. If you create one of each with the same `max_concurrent`, each is enforced separately, so together they admit up to twice that many requests. ### Composition with sync `Client` -The same ordering rationale from [Composition](#composition) applies — `Bulkhead` outside `Retry` — just without `AsyncTimeout` (no sync equivalent) and using `Client`, `Bulkhead`, and `Retry` in place of their async counterparts. +The [composition order](#composition) applies with the sync classes, minus the timeout: `CircuitBreaker`, then `Bulkhead`, then `Retry`. ## See also -- **[Middleware guide](middleware.md)** — write your own resilience middleware against the same protocol `AsyncRetry` and `AsyncBulkhead` use. -- **[Errors reference](errors.md)** — `RetryBudgetExhaustedError`, `BulkheadFullError`, `CircuitOpenError`, and the broader exception tree. -- **[Observability](observability.md)** — the operational events these middleware emit. +- [Middleware](middleware.md): write your own resilience middleware with the same protocol. +- [Errors](errors.md): `RetryBudgetExhaustedError`, `BulkheadFullError`, `CircuitOpenError` and the rest of the exception tree. +- [Observability](observability.md): the events these middleware emit. diff --git a/docs/testing.md b/docs/testing.md index b67d58a..539fe3d 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -1,8 +1,8 @@ # Testing guide -`httpware`'s test seam is `httpx2`. Pass an `httpx2.MockTransport` as `AsyncClient(transport=...)` — the middleware chain still runs end-to-end, only the wire is mocked, and the client still owns and closes its `httpx2` client. No special test mode, no monkey-patching, no `respx`. +To test code that uses httpware, pass an `httpx2.MockTransport` as `AsyncClient(transport=...)`. Only the network is replaced: the middleware chain runs as usual, and the client still creates and closes its `httpx2` client. -A pre-built `httpx2.AsyncClient`/`httpx2.Client` can be passed as `httpx2_client=` instead. It is mutually exclusive with every `httpx2` client option (`base_url`, `headers`, `transport`, `verify`, ...): passing any of them alongside it raises `TypeError`. Configure the client you pass instead; `httpware` will not close it. +You can pass a ready-made `httpx2.AsyncClient` or `httpx2.Client` as `httpx2_client=` instead. It can't be combined with any `httpx2` client option (`base_url`, `headers`, `transport`, `verify`, ...); passing one raises `TypeError`. Configure the client you pass, and close it yourself, since httpware won't. ## The basic pattern @@ -25,13 +25,13 @@ async def test_get_user() -> None: assert response.json()["name"] == "Alice" ``` -The handler can be sync or async; `httpx2.MockTransport` supports both. The test above uses a sync handler. +The handler can be sync, as above, or async. -If you use `pytest-asyncio` in auto-mode (`asyncio_mode = "auto"` under `[tool.pytest.ini_options]`), async test functions don't need the `@pytest.mark.asyncio` decorator. +If you use `pytest-asyncio` in auto mode (`asyncio_mode = "auto"` under `[tool.pytest.ini_options]`), async test functions don't need the `@pytest.mark.asyncio` decorator. ### Sync `Client` -The same pattern works for the sync `Client`; `httpx2.MockTransport` serves both worlds: +The same works for the sync `Client`: ```python from http import HTTPStatus @@ -54,7 +54,7 @@ def test_get_returns_typed_response() -> None: ## Recording / stateful handlers -For tests that need to vary the response by call count or assert on the requests that came in, use a handler with instance state: +To change the response from call to call, or to check the requests that were sent, give the handler some state: ```python from httpware import AsyncRetry @@ -84,11 +84,11 @@ async def test_retry_succeeds_after_503() -> None: assert len(handler.calls) == 2 # initial + 1 retry ``` -The `base_delay`/`max_delay` are set tiny so the test runs instantly — no need for `freezegun` or sleep injection in most cases. +The tiny `base_delay` and `max_delay` keep the test fast, so you usually don't need `freezegun` or a fake sleep. ## Testing your custom middleware -Compose your middleware with the mock transport to exercise the chain end-to-end: +Run your middleware against the mock transport to test it inside the real chain: ```python async def test_my_middleware_adds_header() -> None: @@ -101,14 +101,13 @@ async def test_my_middleware_adds_header() -> None: assert handler.calls[0].headers["X-My-Header"] == "expected-value" ``` -For middleware with state-keeping (counters, circuit-breaker state), assert on instance attributes after running the call. +For middleware that keeps state, such as a counter, assert on its attributes after the call. ## Why not `respx`? -`httpware` deliberately uses `httpx2.MockTransport` instead of `respx` for its own tests. `MockTransport` is a first-party `httpx2` API — supported by the maintainers, stable across versions, part of the public API surface. `respx` targets the original `httpx` package (its README requires `httpx 0.25+`, with no stated `httpx2` support) and has a documented history of breaking across `httpx` major-version bumps, since it patches `httpx`/`httpcore` internals directly. Stick with `MockTransport` unless you have a specific reason not to. +httpware's own tests use `httpx2.MockTransport`, which is part of `httpx2`'s public API. `respx` is built for the original `httpx` package: its README requires `httpx 0.25+` and says nothing about `httpx2`. It also patches `httpx` and `httpcore` internals, which has broken it across `httpx` major versions before. ## See also -- **[Middleware guide](middleware.md)** — write the middleware you're testing. -- **[Resilience reference](resilience.md)** — testing `AsyncRetry`/`AsyncBulkhead` configurations. -- **`AGENTS.md`** — the project's own testing conventions, and the admission check that decides where a new fact belongs. +- [Middleware](middleware.md): writing the middleware you're testing. +- [Resilience](resilience.md): the parameters of the middleware you're configuring. diff --git a/mkdocs.yml b/mkdocs.yml index 5b319de..39f42ff 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -5,12 +5,12 @@ docs_dir: docs edit_uri: edit/main/docs/ nav: - - Quick-Start: index.md + - Quickstart: index.md - Resilience: resilience.md - Demos: - Overview: demos/index.md - Retry: demos/retry.md - - Circuit Breaker: demos/circuit-breaker.md + - Circuit breaker: demos/circuit-breaker.md - Bulkhead: demos/bulkhead.md - Timeout: demos/timeout.md - Full stack: demos/full-stack.md diff --git a/pyproject.toml b/pyproject.toml index 767fd09..ed04261 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "httpware" -description = "Python HTTP client framework with sync & async clients and built-in resilience" +description = "Typed, resilient HTTP clients for Python, sync and async" authors = [{ name = "Artur Shiriev", email = "me@shiriev.ru" }] requires-python = ">=3.11,<4" license = "MIT" diff --git a/src/httpware/middleware/resilience/retry.py b/src/httpware/middleware/resilience/retry.py index 7e9e5d6..f0cd19e 100644 --- a/src/httpware/middleware/resilience/retry.py +++ b/src/httpware/middleware/resilience/retry.py @@ -56,7 +56,9 @@ _RETRYABLE_EXCEPTIONS = (StatusError, NetworkError, TimeoutError) _MAX_ATTEMPTS_INVALID = "max_attempts must be >= 1" -_STREAMING_BODY_REFUSAL_NOTE = "httpware: not retrying — request body is a stream that cannot replay across attempts" +_STREAMING_BODY_REFUSAL_NOTE = ( + "httpware: not retrying because the request body is a stream that cannot replay across attempts" +) _RETRY_AFTER_EXCEEDS_MAX_DELAY_NOTE = ( "httpware: Retry-After ({retry_after}s) exceeded max_delay ({max_delay}s); giving up" )